"""Defaults, types, categories, descriptions, and validation for settings."""
import inspect
import logging
import sys
import os, ast
from copy import deepcopy
from dataclasses import dataclass, replace
from numbers import Integral
from .organelle_types import (ALL_ORGANELLE_ROLES,
DEFAULT_NUMBER_OF_ORGANELLES,
DEFAULT_TYPE as DEFAULT_ORGANELLE_TYPE,
NUMBER_OF_ORGANELLES, active_organelle_roles,
apply_preset, declared_organelle_roles,
organelle_count, organelle_number,
organelle_slot_label,
slot_setting,
_background_switch_key,
_legacy_background_switch_role)
LOG = logging.getLogger(__name__)
#: Every organelle slot that can be DECLARED, which is not the same as the
#: number a given run HAS.
#:
#: The registries below -- types, tooltips, categories, widget kinds -- are
#: generated for all of them, so no slot can reach a panel as a control with
#: no type and no help, and a settings file written at seven slots still
#: loads, validates and is written back out when it is opened at two. How
#: many slots a run actually shows is `number_of_organelles`, read through
#: :func:`spacr.organelle_types.active_organelle_roles`.
ORGANELLE_SLOT_ROLES = ALL_ORGANELLE_ROLES
#: The feature-group value meaning shape measurements rather than an image
#: channel. Kept in this lightweight settings module because a configuration
#: screen must be able to canonicalise its own value without importing the
#: training/image stack in ``spacr.utils``.
FEATURE_SELECTION_MORPHOLOGY = 'morphology'
from . import graph_types as _graph_types
[docs]
def canonical_feature_selection(value):
"""Return ``channel_of_interest`` in the one form spaCR stores.
``None``, an empty value and ``'all'`` mean every feature. One channel is
an integer, several channels are an order-preserving list, morphology is
the literal ``'morphology'``, and other strings are column-name filters.
A one-member collection collapses to its member so the settings panel
cannot turn channel ``3`` into a different results path named ``[3]``.
This function deliberately uses only the standard library. It runs while
configuration widgets are collected, before the user starts a pipeline;
importing numpy/torch/cv2 merely to normalise a combo-box answer made
opening Classify allocate hundreds of megabytes.
:param value: scalar, string, or collection supplied by a settings widget,
CSV, script or default.
:returns: ``None``, an integer, a string, or an order-preserving list.
:raises ValueError: when the value cannot name a feature selection.
"""
if value is None:
return None
if isinstance(value, str):
text = value.strip()
if not text or text.lower() in ('all', 'none'):
return None
if text.lower() == FEATURE_SELECTION_MORPHOLOGY:
return FEATURE_SELECTION_MORPHOLOGY
if ',' in text:
return canonical_feature_selection(
[part for part in text.split(',')])
try:
return int(text)
except ValueError:
return text
if isinstance(value, bool):
raise ValueError(
f"channel_of_interest={value!r} is a boolean; it names no "
"channel. Use a channel number, 'morphology', or None for "
"every feature.")
if isinstance(value, Integral):
return int(value)
if isinstance(value, (list, tuple, set)):
members = [canonical_feature_selection(item) for item in value]
members = [member for member in members if member is not None]
seen, unique = set(), []
for member in members:
key = (type(member).__name__, member)
if key not in seen:
seen.add(key)
unique.append(member)
if not unique:
return None
return unique[0] if len(unique) == 1 else unique
raise ValueError(
f"channel_of_interest={value!r} is a {type(value).__name__}; it "
"must be a channel number, 'morphology', a column-name fragment, a "
"list of those, or None for every feature.")
def _organelle_slot_key(key, role):
"""Translate a primary ``organelle_*`` key to one slot's key."""
return slot_setting(key, role)
def _clone_primary_organelle_values(settings, roles=None):
"""Default every secondary slot from the primary slot's current values.
:param settings: the settings dict, filled in place.
:param roles: the slot prefixes to fill, primary slot included. Defaults
to the slots this dict DECLARES -- the ones `number_of_organelles`
asks for plus any further slot the dict already carries a key for --
so lowering the number hides slots without dropping the answers they
hold.
"""
primary = [(key, deepcopy(value)) for key, value in settings.items()
if str(key).startswith('organelle_')]
if roles is None:
roles = declared_organelle_roles(settings)
for role in tuple(roles)[1:]:
for key, value in primary:
settings.setdefault(_organelle_slot_key(key, role),
deepcopy(value))
return settings
[docs]
def organelle_slots_beyond_the_count(settings, count=None):
"""Populate declarable organelle slots without changing the active count.
Extra slot keys allow the settings interface to reveal controls when
``number_of_organelles`` increases. Existing values take precedence.
:param settings: Run-settings mapping; copied rather than modified.
:param count: Number of slots to populate. ``None`` populates all roles,
bounded by :data:`spacr.organelle_types.MAX_ORGANELLES`.
:returns: New settings dictionary. Mappings without organelle settings
are returned unchanged.
"""
from .organelle_types import MAX_ORGANELLES, organelle_roles
out = dict(settings or {})
if not any(str(key).startswith('organelle_') for key in out):
return out
wanted = MAX_ORGANELLES if count is None else max(int(count), 0)
return _clone_primary_organelle_values(out, organelle_roles(wanted))
def _clone_organelle_registry(mapping, *, tooltip=False):
"""Generate every declarable slot's entries from the primary registry.
Registries are generated for :data:`ORGANELLE_SLOT_ROLES` rather than for
the slots a run happens to have, because a registry is what makes a key
READABLE: a settings file written at seven slots is opened by a session
whose count is two, and its seventh slot has to be typed and tooltipped
then, not only once the number is raised again.
"""
primary = [(key, value) for key, value in mapping.items()
if str(key).startswith('organelle_')]
for role in ORGANELLE_SLOT_ROLES[1:]:
number = organelle_number(role)
for key, value in primary:
cloned = value
if tooltip and isinstance(value, str):
cloned = value.replace('organelle_', f'{role}_')
cloned = cloned.replace('organelle ', f'organelle {number} ')
mapping.setdefault(_organelle_slot_key(key, role), cloned)
return mapping
DEFAULT_BARCODE_REGEX = (
r"^(?P<columnID>.{8})TGCTG.*TAAAC"
r"(?P<grna>.{20,21})AACTT.*AGAAG(?P<rowID>.{8}).*"
)
_BUNDLED_BARCODE_FILES = {
"column": "barcodes_column.csv",
"grna": "barcodes_grna.csv",
"row": "barcodes_row.csv",
}
def _default_worker_count(reserve=0):
"""Return a usable worker default while leaving ``reserve`` CPU cores free.
``os.cpu_count()`` may be ``None`` and small machines or hosted runners
commonly expose only two or four cores. Direct subtraction therefore
made the shipped mask default zero on GitHub Actions, after which spaCR's
own preflight correctly rejected it.
"""
cores = os.cpu_count() or 1
return max(1, int(cores) - max(0, int(reserve)))
[docs]
def bundled_barcode_path(kind):
"""Return the installed CSV path for a bundled barcode reference.
:param kind: ``'column'``, ``'grna'`` or ``'row'``.
:returns: absolute path to the packaged CSV.
:raises ValueError: when ``kind`` is not a bundled reference type.
"""
try:
filename = _BUNDLED_BARCODE_FILES[str(kind).lower()]
except KeyError as exc:
choices = ", ".join(_BUNDLED_BARCODE_FILES)
raise ValueError(
f"Unknown barcode reference {kind!r}; choose {choices}."
) from exc
return os.path.abspath(
os.path.join(os.path.dirname(__file__), "resources", "data", filename)
)
#: sha256 of each bundled barcode CSV, checked before a downloaded copy is
#: trusted.
#:
#: THIS PIN CANNOT GO STALE SILENTLY. `test_the_pinned_barcode_hashes_are_the
#: _bundled_files` recomputes every value from the file that ships beside it,
#: so editing a CSV without editing this table is a red test rather than a
#: download that is rejected on a user's machine months later. That failure
#: mode is not hypothetical here: `tools/build_instruction_index.py` carried a
#: hand-written table nothing re-checked, and items stayed blocked in it for
#: weeks after the thing they waited for had happened.
BUNDLED_BARCODE_SHA256 = {
"column":
"1736196d02c3b32a85a6934fd251dc977d6db6324665ad04dca3193f4ba41063",
"grna":
"0b304fcab3034c3008367406574a6addb91c2adb93dd961a39448c9aea036936",
"row":
"3117fdaf4c551ed9afb8ee4e2e49bd774cab885b125362a58e2d16e99c52a97e",
}
#: The settings key each bundled reference fills, when that key is empty.
BUNDLED_BARCODE_SETTING = {
"column": "column_csv",
"grna": "grna_csv",
"row": "row_csv",
}
def _bundled_barcode_url(kind, version=None):
"""Return the GitHub raw URL for a bundled barcode reference.
:param kind: ``'column'``, ``'grna'`` or ``'row'``.
:param version: the release to fetch from; the installed one by default.
:returns: the raw URL of that CSV at that release's tag.
:raises ValueError: when ``kind`` is not a bundled reference type.
PINNED TO A TAG AND NEVER TO A BRANCH. A URL on `nightly` returns whatever
the file happens to be today, which would not match
:data:`BUNDLED_BARCODE_SHA256` and would be rejected -- correctly, but
with an integrity complaint rather than the truth, which is that the
address was wrong.
"""
if str(kind).lower() not in _BUNDLED_BARCODE_FILES:
choices = ", ".join(_BUNDLED_BARCODE_FILES)
raise ValueError(
f"Unknown barcode reference {kind!r}; choose {choices}.")
filename = _BUNDLED_BARCODE_FILES[str(kind).lower()]
if version is None:
from . import __version__ as version
return (
"https://raw.githubusercontent.com/EinarOlafsson/spacr/"
f"v{version}/spacr/resources/data/{filename}")
def _verified_barcode_bytes(kind, payload):
"""Return ``payload`` if it is the bundled reference, or raise.
:param kind: which bundled reference was asked for.
:param payload: the bytes that came back.
:returns: the same bytes, once they are known to be the right ones.
:raises ValueError: when the bytes are not what this release ships.
"""
import hashlib
expected = BUNDLED_BARCODE_SHA256[str(kind).lower()]
got = hashlib.sha256(payload).hexdigest()
if got != expected:
raise ValueError(
f"The downloaded {kind} barcode reference is not the one this "
f"release ships: expected sha256 {expected}, got {got}. It was "
f"not written. A barcode table that is not the expected one maps "
f"reads to the wrong wells, which produces a finished result "
f"rather than an error.")
return payload
def _ensure_bundled_barcode(kind, fetch=None):
"""Return the path to a bundled barcode reference, fetching it if missing.
:param kind: ``'column'``, ``'grna'`` or ``'row'``.
:param fetch: a callable taking a URL and returning bytes, for tests and
for callers that route their own network access. ``None`` uses
``urllib``.
:returns: the absolute path to the CSV.
:raises ValueError: when ``kind`` is unknown, or the fetched bytes do not
match the hash this release pins.
:raises OSError: when the file is absent and cannot be fetched.
THE NETWORK IS THE LAST RESORT, NOT THE FIRST. These CSVs ship inside the
wheel, so the common path touches no network at all; the fetch exists for
an install whose package data was stripped. Checking the local copy first
also means a machine with no route out still works.
"""
path = bundled_barcode_path(kind)
if os.path.exists(path):
return path
url = _bundled_barcode_url(kind)
if fetch is None:
def fetch(target):
"""Read ``target`` over the network and return its bytes.
:param target: the raw GitHub URL of one bundled CSV.
:returns: the response body, unchecked -- the caller verifies it.
"""
from urllib.request import urlopen
with urlopen(target, timeout=30) as response: # noqa: S310
return response.read()
payload = _verified_barcode_bytes(kind, fetch(url))
os.makedirs(os.path.dirname(path), exist_ok=True)
staging = f"{path}.partial"
with open(staging, "wb") as handle:
handle.write(payload)
os.replace(staging, path)
return path
def _fill_missing_barcode_references(settings, fetch=None):
"""Point every EMPTY barcode reference setting at the bundled table.
:param settings: the settings mapping to fill in place.
:param fetch: passed through to :func:`_ensure_bundled_barcode`.
:returns: the names of the settings that were filled, in order.
A REFERENCE THE USER SET IS NEVER OVERWRITTEN, which is the whole
contract. Loading test data is something a user does part-way through
setting a run up, and silently replacing the guide table they just chose
with the bundled one would be a mapping run against the wrong references
that completes and reports numbers.
A reference that cannot be produced is LEFT EMPTY rather than raising.
The other two are still worth filling, and an empty field is a state the
screen already knows how to show; a raised exception out of "load test
data" is not.
"""
filled = []
for kind, key in BUNDLED_BARCODE_SETTING.items():
if str(settings.get(key) or "").strip():
continue
try:
settings[key] = _ensure_bundled_barcode(kind, fetch=fetch)
except (OSError, ValueError):
LOG.info("no bundled %s barcode reference to fill in", kind,
exc_info=True)
continue
filled.append(key)
return filled
@dataclass(frozen=True)
[docs]
class BarcodeEntry:
"""One barcode type that a run decodes.
A run used to decode exactly three barcodes -- a plate column, a guide
and a plate row -- because the regex, the settings and the read
processors each spelled all three of them out by name. An entry is that
same information written once, so that a screen carrying a fourth
barcode, or only one, is a different collection of entries rather than a
different code path.
:ivar name: the word a user would use for this barcode, such as
``column``, ``row``, ``grna`` or ``plate``. It names the two output
columns the run writes for the entry, so it has to be unique within a
set.
:ivar csv: the reference table that turns one of these sequences into a
name. It needs a ``sequence`` column and a ``name`` column. Its
sequences must be in the same orientation as the reads, because they
are compared verbatim rather than reverse-complemented.
:ivar group: the named group of the barcode regex whose captured text is
this barcode. Left empty it is the entry's own name, which is what a
regex written for a new set will normally use.
:ivar group_aliases: further spellings of that group name, accepted when
the preferred one is absent from the regex. The column and row
barcodes spaCR shipped accept the shorter ``column`` and ``row`` this
way, which is why a pattern written before those names were settled
still runs.
:ivar sequence_column: the output column the extracted sequence is written
to. Left empty it is the entry's name followed by ``_sequence``.
:ivar id_column: the output column the resolved name is written to. Left
empty it is the entry's name followed by ``ID``. The guide barcode
spaCR shipped sets this to ``grna_name`` instead, because that is the
header every count table already written and every reader of one
expects.
"""
name: str
csv: str = ""
group: str = ""
group_aliases: tuple = ()
sequence_column: str = ""
id_column: str = ""
[docs]
def __post_init__(self):
"""Fill in the spellings that follow from the entry's own name.
Derived here rather than at each point of use, because two readers
that each work out what an entry's columns are called are two
readers that can disagree. The entry is frozen, so the normalised
values are assigned through ``object.__setattr__``.
:returns: None.
:raises ValueError: when the entry has no name, which would leave
its output columns called nothing at all.
"""
name = str(self.name or "").strip()
if not name:
raise ValueError(
"A barcode entry needs a name: it names the columns the run "
"writes for that barcode and identifies it in every message.")
object.__setattr__(self, "name", name)
object.__setattr__(self, "csv", str(self.csv or ""))
object.__setattr__(self, "group", str(self.group or "").strip() or name)
object.__setattr__(self, "group_aliases", tuple(
str(alias).strip() for alias in (self.group_aliases or ())
if str(alias).strip()))
object.__setattr__(self, "sequence_column",
str(self.sequence_column or "").strip()
or f"{name}_sequence")
object.__setattr__(self, "id_column",
str(self.id_column or "").strip() or f"{name}ID")
[docs]
def accepted_groups(self):
"""Return every regex group name this barcode answers to.
:returns: a tuple of group names, the preferred spelling first and
any older accepted spellings after it.
"""
return (self.group,) + tuple(
alias for alias in self.group_aliases if alias != self.group)
[docs]
def group_in(self, names):
"""Return the spelling of this barcode's group that a regex uses.
:param names: the group names a regex defines.
:returns: the accepted group name the regex defines, or None when it
defines none of them.
"""
available = set(names)
for candidate in self.accepted_groups():
if candidate in available:
return candidate
return None
[docs]
def group_label(self):
"""Return this barcode's accepted group names as one label.
Sorted rather than preferred first, because this label lands in the
message a user reads when the regex is missing a group, and the
message spaCR has always printed names the older spelling first.
:returns: the accepted group names joined by slashes.
"""
return "/".join(sorted(set(self.accepted_groups())))
@dataclass(frozen=True)
[docs]
class BarcodeSet:
"""The barcode types a run decodes, in order.
A set replaces the three named reference settings the module started
with. Iterating it is how the run reaches every barcode, so a run with
one barcode and a run with ten differ only in what this holds.
The count columns can be set separately because the three barcodes spaCR
shipped for years are counted by row, then column, then guide, while the
reads list the column first. A set reproducing that run has to be able to
say so, rather than quietly re-sort tables people already have.
:ivar entries: the barcode types, in the order the run lists them. Every
entry adds a sequence column and a name column to each annotated
read. Those columns come after the read, in the same order.
:ivar count_columns: the name columns the per-well counts are grouped by,
in the order they are grouped. Left empty it is every entry's name
column in entry order.
"""
entries: tuple = ()
count_columns: tuple = ()
[docs]
def __post_init__(self):
"""Normalise the collection and refuse one that cannot decode a read.
Every check here is a way for two barcodes to become
indistinguishable in the output, which is silent rather than loud:
two entries sharing a name column overwrite each other in the
annotated reads, and counts grouped by the wrong columns are still a
table full of plausible numbers.
:returns: None.
:raises ValueError: when the set is empty, when a member is not a
barcode entry, when two entries share a name, a regex group or
an output column, or when the count columns are not exactly the
entries' name columns.
"""
entries = tuple(self.entries or ())
if not entries:
raise ValueError(
"A barcode set needs at least one entry; a run with no "
"barcode to decode has nothing to count.")
for entry in entries:
if not isinstance(entry, BarcodeEntry):
raise ValueError(
"A barcode set holds BarcodeEntry values; received "
f"{type(entry).__name__}. Use barcode_set_from_settings "
"to build a set from a settings file.")
object.__setattr__(self, "entries", entries)
for label, values in (
("name", [entry.name for entry in entries]),
("regex group", [entry.group for entry in entries]),
("sequence column",
[entry.sequence_column for entry in entries]),
("name column", [entry.id_column for entry in entries])):
repeated = sorted({value for value in values
if values.count(value) > 1})
if repeated:
raise ValueError(
f"Two barcodes in this set share a {label}: "
f"{', '.join(repeated)}. Each barcode needs its own, or "
"one of them silently replaces the other in the output.")
wanted = tuple(entry.id_column for entry in entries)
given = tuple(str(column) for column in (self.count_columns or ()))
if not given:
given = wanted
elif sorted(given) != sorted(wanted):
raise ValueError(
"The count columns of this barcode set are "
f"{', '.join(given)}, which is not its barcodes' name "
f"columns, {', '.join(wanted)}. Counts are per unique "
"combination of every barcode, so the two hold the same "
"columns and differ only in order.")
object.__setattr__(self, "count_columns", given)
[docs]
def __len__(self):
"""Return how many barcodes this set decodes.
:returns: the number of entries as an integer.
"""
return len(self.entries)
[docs]
def __iter__(self):
"""Iterate the barcodes in the order the run lists them.
:returns: an iterator over the entries.
"""
return iter(self.entries)
@property
[docs]
def names(self):
"""Return the name of each barcode, in entry order.
:returns: a tuple of names.
"""
return tuple(entry.name for entry in self.entries)
@property
[docs]
def id_columns(self):
"""Return the output column each barcode's resolved name lands in.
:returns: a tuple of column names, in entry order.
"""
return tuple(entry.id_column for entry in self.entries)
@property
[docs]
def sequence_columns(self):
"""Return the output column each barcode's raw sequence lands in.
:returns: a tuple of column names, in entry order.
"""
return tuple(entry.sequence_column for entry in self.entries)
[docs]
def resolve_groups(self, regex):
"""Return the regex group each barcode is captured by.
A set of five barcodes needs five named groups, and the failure this
answers is a regex that names four. That used to surface as a bare
"no such group" from inside a worker process, several frames from
anything a user configured, so the message here names the barcode
that has no group and then lists the groups the regex does define.
:param regex: a compiled regular expression, or the pattern string
of one.
:returns: a dict from each barcode's name to the group name the
regex spells it with.
:raises ValueError: when the regex names no group for one or more of
the barcodes in this set.
"""
import re as _re
compiled = regex if hasattr(regex, "groupindex") else _re.compile(regex)
available = set(compiled.groupindex)
resolved, missing = {}, []
for entry in self.entries:
found = entry.group_in(available)
if found is None:
missing.append(entry.group_label())
else:
resolved[entry.name] = found
if missing:
message = ("Barcode regex is missing required named group(s): "
+ ", ".join(missing) + ".")
if available:
message += (" The groups this regex defines are "
+ ", ".join(sorted(available)) + ".")
else:
message += " This regex defines no named groups at all."
raise ValueError(message)
taken = {}
for name, group in resolved.items():
if group in taken:
raise ValueError(
f"The {taken[group]} and {name} barcodes would both be "
f"read from the regex group {group}, so both would be "
"given the same sequence. Give one of them a group of "
"its own.")
taken[group] = name
return resolved
#: Private like `_BUNDLED_BARCODE_FILES` beside it, and reached the same
#: way: through the function that uses it rather than by importing it.
#:
#: What the three barcodes spaCR has shipped since before barcode sets
#: existed are called, so that a settings file naming one of them by name
#: gets the run it has always got rather than a generically derived one.
#:
#: ONLY THE FIELDS A SETTINGS FILE LEAVES OUT ARE FILLED FROM THIS. An entry
#: that spells a field out keeps what it says, and a name that is not one of
#: these three derives its spellings from itself the ordinary way.
#:
#: The guide is the one that cannot be derived: its resolved name lands in
#: `grna_name` rather than `grnaID`, which is the header every count table
#: already written uses and every reader of one, `spacr.ml` included,
#: expects. The column and row prefer the longer group name and still accept
#: the short one, which is how a regex written before those names were
#: settled goes on matching.
_SHIPPED_BARCODE_SPELLINGS = {
'column': {'group': 'columnID', 'group_aliases': ('column',),
'id_column': 'columnID'},
'row': {'group': 'rowID', 'group_aliases': ('row',),
'id_column': 'rowID'},
'grna': {'group': 'grna', 'id_column': 'grna_name'},
}
#: The order the per-well counts of the three shipped barcodes are grouped
#: in, which is not the order the reads list them in. Both orders are older
#: than barcode sets and both are in files people already have.
_SHIPPED_COUNT_COLUMNS = ('rowID', 'columnID', 'grna_name')
[docs]
def barcode_set_from_settings(settings):
"""Return the barcode set a Map Barcodes run was configured with.
A settings file names its barcodes under ``barcode_set``, as a list with
one entry per barcode. An entry is a mapping of the fields of
:class:`BarcodeEntry`, or just a name when the reference table for that
name is already in the settings. An entry that names no reference table
takes the one under its own name followed by ``_csv``, and failing that
the reference of that name spaCR ships, so adding a fourth barcode to
the three that ship is a one-entry addition rather than a re-declaration
of all four.
An entry named after one of the three barcodes spaCR shipped keeps the
spellings that barcode has always had, for every field the settings file
does not spell out itself. That is what makes a set of those three the
run they describe rather than a re-implementation of it: the same regex
groups, the same output columns, and the same order of the count table.
A barcode named anything else derives its spellings from its own name.
:param settings: a Map Barcodes settings mapping.
:returns: the configured :class:`BarcodeSet`, or None when the settings
name no set. None is not an error and is the ordinary case: it means
the run decodes the plate column, the guide and the plate row named
by ``column_csv``, ``grna_csv`` and ``row_csv``, exactly as every
run did before a set could be named at all.
:raises ValueError: when an entry is neither a name nor a mapping of
entry fields, when a mapping carries a field no entry has, or when
an entry names a reference table that cannot be resolved.
"""
if not isinstance(settings, dict):
return None
value = settings.get('barcode_set')
if isinstance(value, BarcodeSet):
return value
if not value:
return None
count_columns = ()
if isinstance(value, dict):
count_columns = tuple(value.get('count_columns') or ())
value = value.get('entries') or ()
if isinstance(value, (str, bytes)) or not hasattr(value, '__iter__'):
raise ValueError(
"barcode_set is a list with one entry per barcode; received "
f"{type(value).__name__}.")
fields = {'name', 'csv', 'group', 'group_aliases', 'sequence_column',
'id_column'}
entries = []
for item in value:
if isinstance(item, BarcodeEntry):
entry = item
else:
if isinstance(item, str):
values = {'name': item}
elif isinstance(item, dict):
unknown = sorted(set(map(str, item)) - fields)
if unknown:
raise ValueError(
f"A barcode_set entry names {', '.join(unknown)}, "
f"which a barcode has no field for. The fields are "
f"{', '.join(sorted(fields))}.")
values = {str(key): item[key] for key in item}
else:
raise ValueError(
"A barcode_set entry is the barcode's name or a mapping "
f"of its fields; received {type(item).__name__}.")
for field, spelling in _SHIPPED_BARCODE_SPELLINGS.get(
str(values.get('name', '')), {}).items():
values.setdefault(field, spelling)
entry = BarcodeEntry(**values)
if not entry.csv:
fallback = settings.get(f'{entry.name}_csv')
if not fallback and entry.name.lower() in _BUNDLED_BARCODE_FILES:
fallback = bundled_barcode_path(entry.name)
if not fallback:
raise ValueError(
f"The {entry.name} barcode has no reference table. Give "
f"the entry a csv, or set {entry.name}_csv beside the "
"set.")
entry = replace(entry, csv=str(fallback))
entries.append(entry)
entries = tuple(entries)
if not count_columns and sorted(
entry.id_column for entry in entries) == sorted(
_SHIPPED_COUNT_COLUMNS):
count_columns = _SHIPPED_COUNT_COLUMNS
return BarcodeSet(entries, count_columns=count_columns)
#: app key → defaults factory. Written by :func:`register_defaults`.
_DEFAULTS_REGISTRY = {}
#: Category headings contributed by a module through
#: :func:`register_defaults`, rather than typed into the `categories`
#: literal further down. Power/Design is the one that exists today.
REGISTERED_CATEGORIES = set()
[docs]
def register_defaults(app_key, fn, *, replace=False, expected_types=None,
tooltips=None, categories=None, description=None):
"""Register the defaults factory for ``app_key``.
:param app_key: the app key the factory belongs to — the same key the
module registered with :func:`spacr.qt.app.register_app`.
:param fn: callable returning the module's settings dict. Called as
``fn(settings)`` when it takes an argument and ``fn()`` when it
does not, so both existing shapes in this file work unchanged.
:param replace: allow overwriting an existing registration. Off by
default: two modules quietly claiming one key is the failure this
registry exists to make loud.
:param expected_types: optional ``{key: type-or-tuple}`` merged into
:data:`expected_types`. Without it a new key is untyped and
:func:`check_settings` cannot validate it.
:param tooltips: optional ``{key: text}`` merged into
:data:`tooltips`. Without it the settings panel has no help for
the key, which the GUI suite fails on.
:param categories: optional ``{category: [key, ...]}`` merged into
:data:`categories`; unknown category names are created. A key
already in that category is not added twice.
:param description: optional module blurb for :data:`descriptions`.
:raises ValueError: on a re-registration without ``replace``, or when
a merged type/tooltip/description would change one that already
exists — a module may add to the shared tables, never rewrite
another module's entry.
:raises TypeError: if ``fn`` is not callable.
"""
app_key = str(app_key)
if not app_key:
raise ValueError("defaults need an app key")
if not callable(fn):
raise TypeError(f"defaults for {app_key!r} are not callable: {fn!r}")
if app_key in _DEFAULTS_REGISTRY and not replace:
raise ValueError(
f"defaults for {app_key!r} are already registered; pass "
"replace=True if that is really what you mean")
_merge_declarations(app_key, expected_types, tooltips, categories,
description)
_DEFAULTS_REGISTRY[app_key] = fn
return fn
def _merge_declarations(app_key, types_, tips, cats, description):
"""Fold a module's type/tooltip/category/description contributions in.
Done before the factory is stored so a rejected contribution leaves
the registry untouched — a half-registered module whose keys are
typed but whose defaults are missing is harder to diagnose than one
that failed at import.
"""
for key, type_ in dict(types_ or {}).items():
existing = expected_types.get(key, type_)
if existing != type_:
raise ValueError(
f"{app_key!r} declares {key!r} as {type_!r}, but it is "
f"already declared as {existing!r}")
expected_types[key] = type_
for key, text in dict(tips or {}).items():
existing = tooltips.get(key, text)
if existing != text:
raise ValueError(
f"{app_key!r} redefines the tooltip for {key!r}; extend the "
"existing text instead of replacing another module's help")
tooltips[key] = text
for name, keys in dict(cats or {}).items():
new_category = name not in categories
bucket = categories.setdefault(name, [])
for key in keys:
if key not in bucket:
bucket.append(key)
if new_category and name not in category_keys:
category_keys.append(name)
if new_category:
REGISTERED_CATEGORIES.add(name)
if description is not None:
existing = descriptions.get(app_key, description)
if existing != description:
raise ValueError(
f"{app_key!r} already has a description; edit that one "
"rather than registering a second")
descriptions[app_key] = description
[docs]
def unregister_defaults(app_key):
"""Drop a registered defaults factory. ``True`` if there was one.
Only the factory: the types, tooltips and categories it merged stay,
because another module may already have added keys to the same
category and unpicking a merge is guesswork.
:param app_key: key to drop, coerced with ``str()`` exactly as
:func:`register_defaults` stored it. An unknown key is not an
error — the call returns ``False`` — so a test teardown can run
unconditionally. Dropping a key first is also how a module
re-registers without reaching for ``replace=True``.
"""
return _DEFAULTS_REGISTRY.pop(str(app_key), None) is not None
[docs]
def has_registered_defaults(app_key):
"""Whether a module registered a defaults factory for ``app_key``.
:param app_key: key to look up, coerced with ``str()`` as
:func:`register_defaults` stored it. This registry holds only what
registers itself, so the built-in ``set_default_*`` families in
this file answer ``False`` — they are reached through the GUI
dispatch instead. ``True`` only promises a factory is there, not
that it works: :func:`defaults_for` still has to run it.
"""
return str(app_key) in _DEFAULTS_REGISTRY
[docs]
def registered_default_apps():
"""Every app key with registered defaults, in registration order."""
return tuple(_DEFAULTS_REGISTRY)
[docs]
def defaults_for(app_key, settings=None):
"""The registered defaults dict for ``app_key``.
The read side of :func:`register_defaults`, and what a settings panel
calls. Returns a fresh dict every time: the caller edits what it gets
back, and a factory that hands out one shared dict would let one
module's screen edit another's defaults.
:param app_key: key some module passed to :func:`register_defaults`.
Only that registry is consulted, so the ``set_default_*`` families
defined below in this file are not reachable through it; the error
lists the keys that are.
:param settings: values to seed the factory with, exactly like the
``settings`` argument of every ``set_default_*`` in this file.
:raises KeyError: when nothing is registered for ``app_key``.
:raises TypeError: when the factory does not return a dict.
"""
app_key = str(app_key)
try:
fn = _DEFAULTS_REGISTRY[app_key]
except KeyError:
raise KeyError(
f"no defaults registered for {app_key!r}; known: "
f"{', '.join(_DEFAULTS_REGISTRY) or '(none)'}") from None
result = fn(dict(settings or {})) if _takes_an_argument(fn) else fn()
if not isinstance(result, dict):
raise TypeError(
f"defaults for {app_key!r} returned {type(result).__name__}, "
"expected dict")
return dict(result)
def _takes_an_argument(fn):
"""Whether ``fn`` accepts a positional settings dict.
Signature inspection rather than "call it and retry on TypeError":
a retry cannot tell a wrong call from a TypeError raised *inside* a
factory that was called correctly, and would run half of it twice.
"""
try:
params = inspect.signature(fn).parameters
except (TypeError, ValueError):
return True
for param in params.values():
if param.kind in (inspect.Parameter.POSITIONAL_ONLY,
inspect.Parameter.POSITIONAL_OR_KEYWORD,
inspect.Parameter.VAR_POSITIONAL):
return True
return False
[docs]
def set_default_plot_merge_settings():
"""Return the default settings dict for plotting merged mask overlays.
:returns: dict populated with the default ``plot_merge`` parameters
(channel dimensions, backgrounds, overlay behaviour, colormap, etc.).
"""
settings = {}
settings.setdefault('pathogen_limit', 10)
settings.setdefault('nuclei_limit', 1)
settings.setdefault('remove_background', False)
settings.setdefault('filter_min_max', None)
settings.setdefault('channel_dims', [0,1,2,3])
settings.setdefault('backgrounds', [100,100,100,100])
settings.setdefault('cell_mask_dim', 4)
settings.setdefault('nucleus_mask_dim', 5)
settings.setdefault('pathogen_mask_dim', 6)
settings.setdefault('outline_thickness', 3)
settings.setdefault('outline_color', 'gbr')
settings.setdefault('overlay_chans', [1,2,3])
settings.setdefault('overlay', True)
settings.setdefault('normalization_percentiles', [2,98])
settings.setdefault('normalize', True)
settings.setdefault('print_object_number', True)
settings.setdefault('nr', 1)
settings.setdefault('figuresize', 10)
settings.setdefault('cmap', 'inferno')
settings.setdefault('verbose', True)
return settings
def _set_psf_defaults(settings):
"""Populate dormant calibrated PSF settings without loading imaging code."""
settings.setdefault('psf_operation', 'none')
settings.setdefault('psf_source', 'gaussian')
settings.setdefault('psf_path', None)
settings.setdefault('psf_image_sampling_um', None)
settings.setdefault('psf_kernel_sampling_um', None)
settings.setdefault('psf_fwhm_um', None)
settings.setdefault('psf_iterations', 20)
def _set_unmix_defaults(settings):
"""Populate the dormant spectral-unmixing settings, unmixing off."""
settings.setdefault('unmix', False)
settings.setdefault('unmix_controls', '')
settings.setdefault('unmix_background_percentile', 5.0)
def _set_n2v_defaults(settings):
"""Populate the dormant Noise2Void settings, denoising off."""
settings.setdefault('n2v_denoise', False)
settings.setdefault('n2v_model', '')
settings.setdefault('n2v_epochs', 20)
def _set_enhancement_defaults(settings):
"""Populate the dormant enhancement-chain settings, every step off.
One ``enhance_<field>`` per :data:`spacr.qt.detect_chain.SETTINGS_FIELDS`
entry, with the value :data:`spacr.qt.detect_chain.NO_CHAIN` holds --
pinned to it by test rather than imported, so that importing this module
costs no imaging code. Read back by :func:`spacr.psf_pipeline.prepare_chain`.
"""
settings.setdefault('enhance_background', 'none')
settings.setdefault('enhance_background_radius', 50)
settings.setdefault('enhance_background_scale', 0.5)
settings.setdefault('enhance_denoise', 'none')
settings.setdefault('enhance_denoise_strength', 1.0)
settings.setdefault('enhance_percentile_clip', False)
settings.setdefault('enhance_percentile_low', 1.0)
settings.setdefault('enhance_percentile_high', 99.0)
settings.setdefault('enhance_gamma', 1.0)
settings.setdefault('enhance_log', False)
settings.setdefault('enhance_log_gain', 10.0)
settings.setdefault('enhance_sqrt', False)
settings.setdefault('enhance_clahe', False)
settings.setdefault('enhance_clahe_tile', 64)
settings.setdefault('enhance_clahe_clip', 0.01)
settings.setdefault('enhance_equalize', False)
settings.setdefault('enhance_sharpen', False)
settings.setdefault('enhance_sharpen_radius', 1.0)
settings.setdefault('enhance_sharpen_amount', 1.0)
[docs]
def set_default_settings_preprocess_generate_masks(settings=None):
"""Populate default settings for the preprocess/generate-masks pipeline.
Fills channel, Cellpose, plot, timelapse, organelle and post-processing
parameters used by ``preprocess_generate_masks``.
:param settings: optional dict to fill in place; a new dict is created if None.
:returns: the settings dict with defaults applied.
"""
if settings is None:
settings = {}
_fold_renamed_settings(settings)
settings.setdefault('pipeline_style', 'v1')
_set_psf_defaults(settings)
settings.setdefault('psf_objective', 'auto')
_set_unmix_defaults(settings)
_set_n2v_defaults(settings)
_set_enhancement_defaults(settings)
from .image_quality import DEFAULTS as image_quality_defaults
for key, value in image_quality_defaults.items():
settings.setdefault(key, value.copy() if isinstance(value, (dict, list)) else value)
settings.setdefault('segmentation_backend', 'cellpose')
settings.setdefault('batch_fields', 8)
settings.setdefault('keep_npz', False)
settings.setdefault('src', 'path')
settings.setdefault('delete_intermediate', False)
settings.setdefault('preprocess', True)
settings.setdefault('masks', True)
settings.setdefault('save', True)
settings.setdefault('consolidate', False)
settings.setdefault('batch_size', 50)
settings.setdefault('test_mode', False)
settings.setdefault('dry_run', False)
settings.setdefault('test_images', 10)
settings.setdefault('magnification', 40)
settings.setdefault('custom_regex', None)
settings.setdefault('metadata_type', 'cellvoyager')
settings.setdefault('n_jobs', _default_worker_count(reserve=4))
settings.setdefault('ram_guard', True)
settings.setdefault('randomize', True)
settings.setdefault('verbose', True)
settings.setdefault('remove_background_cell', False)
settings.setdefault('remove_background_nucleus', False)
settings.setdefault('remove_background_pathogen', True)
settings.setdefault('remove_background_organelle', False)
settings.setdefault('cell_diameter', None)
settings.setdefault('nucleus_diameter', None)
settings.setdefault('pathogen_diameter', None)
settings.setdefault('diameter_estimate_n_fields', 5)
settings.setdefault('cell_model_name', 'cpsam')
settings.setdefault('nucleus_model_name', 'cpsam')
settings.setdefault('pathogen_model_name', 'cpsam')
settings.setdefault('cellpose3_add_nucleus_channel', True)
settings.setdefault('cellpose3_size_model', False)
settings.setdefault('cellpose3_resample', True)
settings.setdefault('cellpose3_augment', False)
settings.setdefault('cellpose3_percentile_low', 1.0)
settings.setdefault('cellpose3_percentile_high', 99.0)
settings.setdefault('seg_qc', 'report')
settings.setdefault('robustness_report', False)
settings.setdefault('robustness_fields', 4)
settings.setdefault('robustness_crop', 512)
settings.setdefault('robustness_diameter_factors', [0.75, 1.25])
settings.setdefault('robustness_flow_thresholds', [0.2, 0.6])
settings.setdefault('robustness_cellprob_thresholds', [-2.0, 2.0])
settings.setdefault('robustness_enhancement', True)
settings.setdefault('robustness_tolerance', 0.2)
settings.setdefault('real_object_classifier', None)
settings.setdefault('real_object_threshold', 0.5)
settings.setdefault('seg_qc_min_objects', 10)
settings.setdefault('seg_qc_count_ratio', 0.25)
settings.setdefault('seg_qc_size_ratio', 1.4)
settings.setdefault('seg_qc_border_fraction', 0.3)
settings.setdefault('seg_qc_outlier_mad', 5.0)
settings.setdefault('seg_qc_outlier_fraction', 0.15)
settings.setdefault('seg_qc_foreground_fraction', 0.35)
settings.setdefault('seg_qc_split_ratio', 2.0)
settings.setdefault('seg_qc_min_diameter', 5.0)
settings.setdefault('seg_qc_tiny_fraction', 0.3)
settings.setdefault('seg_qc_max_object_fraction', 0.25)
settings.setdefault('seg_qc_plate_fail_fraction', 0.1)
settings.setdefault('cell_channel', None)
settings.setdefault('nucleus_channel', None)
settings.setdefault('pathogen_channel', None)
settings.setdefault('channels', [0,1,2,3])
settings.setdefault('pathogen_background', 200)
settings.setdefault('pathogen_signal_to_noise', 20)
settings.setdefault('pathogen_cellprob_threshold', -1)
settings.setdefault('cell_background', 100)
settings.setdefault('cell_signal_to_noise', 10)
settings.setdefault('cell_cellprob_threshold', 0)
settings.setdefault('nucleus_background', 100)
settings.setdefault('nucleus_signal_to_noise', 10)
settings.setdefault('nucleus_cellprob_threshold', 0)
settings.setdefault('nucleus_flow_threshold', 0.4)
settings.setdefault('cell_flow_threshold', 0.4)
settings.setdefault('pathogen_flow_threshold', 0.4)
settings.setdefault('plot', False)
settings.setdefault('figuresize', 10)
settings.setdefault('cmap', 'inferno')
settings.setdefault('normalize', True)
settings.setdefault('examples_to_plot', 1)
settings.setdefault('pathogen_model', None)
settings.setdefault('merge_pathogens', True)
settings.setdefault('filter', False)
settings.setdefault('lower_percentile', 2)
settings.setdefault('timelapse', False)
settings.setdefault('fps', 2)
settings.setdefault('timelapse_displacement', None)
settings.setdefault('timelapse_memory', 3)
settings.setdefault('timelapse_frame_limits', [5,])
settings.setdefault('timelapse_remove_transient', False)
settings.setdefault('timelapse_mode', 'trackastra')
settings.setdefault('trackastra_model', 'general_2d')
settings.setdefault('trackastra_linking', 'greedy')
settings.setdefault('ultrack_max_distance', 25.0)
settings.setdefault('ultrack_division_weight', -0.1)
settings.setdefault('ultrack_contour_sigma', 0.0)
settings.setdefault('ultrack_n_workers', 1)
settings.setdefault('timeflows_model', None)
settings.setdefault('timelapse_objects', ['cell'])
settings.setdefault('timelapse_lineage', False)
settings.setdefault('timelapse_lineage_color_by', 'generation_time')
settings.setdefault('timelapse_lineage_max_distance', 30.0)
settings.setdefault('timelapse_lineage_min_division_h', 6.0)
settings.setdefault('timelapse_events', False)
settings.setdefault('timelapse_events_annotations', None)
settings.setdefault('timelapse_events_model', None)
settings.setdefault('timelapse_events_window', 9)
settings.setdefault('timelapse_events_threshold', 0.5)
settings.setdefault('timelapse_events_conditions', None)
settings.setdefault('timelapse_events_encoder', 'small')
settings.setdefault('timelapse_events_video_checkpoint', None)
settings.setdefault('timelapse_events_video_channels', None)
settings.setdefault('timelapse_events_video_device', 'auto')
settings.setdefault('save_original_images', True)
settings.setdefault('keep_intermediate', False)
settings.setdefault('keep_original_images', False)
settings.setdefault('adjust_cells', True)
settings.setdefault('mask_parallel', False)
settings.setdefault('mask_gpu_indices', '')
settings.setdefault('watch_folder', False)
settings.setdefault('watch_pipeline', 'mask')
settings.setdefault('watch_normalization_pool', 'per_field')
settings.setdefault('watch_measure_settings', '')
settings.setdefault('watch_classify_settings', '')
settings.setdefault('watch_settle_seconds', 10.0)
settings.setdefault('watch_poll_seconds', 5.0)
settings.setdefault('watch_idle_minutes', 0.0)
settings.setdefault('microscope_feedback', False)
settings.setdefault('microscope_driver', 'simulated')
settings.setdefault('microscope_simulated_folder', '')
settings.setdefault('microscope_positions', '')
settings.setdefault('microscope_stage_transform', [1.0, 0.0, 0.0, 1.0])
settings.setdefault('microscope_event_table', 'cell')
settings.setdefault('microscope_event_query', '')
settings.setdefault('microscope_max_events', 10)
settings.setdefault('microscope_timepoints', 1)
settings.setdefault('microscope_interval_seconds', 0.0)
settings.setdefault('cloud_anonymous', False)
settings.setdefault('cloud_profile', '')
settings.setdefault('cloud_endpoint', '')
settings.setdefault('cloud_cache', '')
settings.setdefault('cloud_wells', '')
settings.setdefault('cloud_fields', 0)
settings.setdefault('cloud_level', 0)
settings.setdefault('cloud_results', '')
settings.setdefault('z_stack', False)
settings.setdefault('z_segmentation_mode', 'project')
settings.setdefault('z_axis', None)
settings.setdefault('z_projection', 'max')
settings.setdefault('anisotropy', None)
settings.setdefault('voxel_size_z_um', None)
settings.setdefault('voxel_size_xy_um', None)
settings.setdefault('stitch_threshold', 0.25)
settings.setdefault('t_stack', False)
settings.setdefault('t_axis_order', None)
settings.setdefault('t_axis', None)
settings.setdefault('frame_interval_s', None)
settings.setdefault('t_track_backend', 'iou')
settings.setdefault('t_link_threshold', 0.25)
settings.setdefault('t_max_displacement_px', None)
settings.setdefault('t_max_displacement_um', None)
settings.setdefault('t_project_for_tracking', False)
_set_organelle_defaults(settings)
settings.setdefault('summarize_organelles_by', 'cell')
settings.setdefault('cell_perimeter_fraction', 0)
settings.setdefault('nucleus_perimeter_fraction', 0)
settings.setdefault('pathogen_perimeter_fraction', 0)
settings.setdefault('organelle_perimeter_fraction', 0)
settings.setdefault('object_filters', {})
_fold_object_bounds(settings)
settings.setdefault('cell_remove_border_objects', False)
settings.setdefault('nucleus_remove_border_objects', False)
settings.setdefault('pathogen_remove_border_objects', False)
settings.setdefault('organelle_remove_border_objects', False)
_clone_primary_organelle_values(settings)
settings.setdefault('motility_analysis', False)
settings.setdefault('strict_errors', None)
settings.setdefault('max_failure_rate', None)
settings.setdefault('resume', False)
from .illumination import illumination_settings
for _key, _value in illumination_settings({}).items():
if _key.startswith('illumination_'):
settings.setdefault(_key, _value)
return settings
[docs]
def get_timelapse_settings(settings=None):
"""Return default settings for the standalone Timelapse module.
The Timelapse module is mask generation run over a time series: the same
preprocessing + Cellpose segmentation as the Mask module, followed by
frame-to-frame linking of the objects named in ``timelapse_objects`` and
per-channel movie export. It therefore takes the full
``set_default_settings_preprocess_generate_masks`` dict with ``timelapse``
forced on — the flag is what the module *is*, not something to configure.
:param settings: optional dict to fill in place; a new dict is created if None.
:returns: the settings dict with defaults applied and ``timelapse`` True.
"""
if settings is None:
settings = {}
settings = set_default_settings_preprocess_generate_masks(settings)
settings['timelapse'] = True
return settings
[docs]
def set_default_plot_data_from_db(settings):
"""Populate default settings for plotting data pulled from a measurements DB.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src', 'path')
settings.setdefault('database', 'measurements.db')
settings.setdefault('graph_name', 'Figure_1')
settings.setdefault('table_names', ['cell', 'cytoplasm', 'nucleus', 'pathogen'])
settings.setdefault('data_column', 'recruitment')
settings.setdefault('grouping_column', 'condition')
settings.setdefault('cell_types', ['Hela'])
settings.setdefault('cell_plate_metadata', None)
settings.setdefault('pathogen_types', None)
settings.setdefault('pathogen_plate_metadata', None)
settings.setdefault('treatments', None)
settings.setdefault('treatment_plate_metadata', None)
settings.setdefault(
'graph_type',
_graph_types.mark_to_start_on(
'categorical_continuous', 'jitter_box')[0])
settings.setdefault('theme', 'deep')
settings.setdefault('save', True)
settings.setdefault('y_lim', None)
settings.setdefault('verbose', False)
settings.setdefault('channel_of_interest', 1)
settings.setdefault('nuclei_limit', 2)
settings.setdefault('pathogen_limit', 3)
settings.setdefault('representation', 'well')
settings.setdefault('uninfected', False)
return settings
[docs]
def set_default_settings_preprocess_img_data(settings):
"""Populate default settings for the image-preprocessing step.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('metadata_type', 'cellvoyager')
settings.setdefault('custom_regex', None)
settings.setdefault('nr', 1)
settings.setdefault('plot', True)
settings.setdefault('batch_size', 50)
settings.setdefault('timelapse', False)
settings.setdefault('lower_percentile', 2)
settings.setdefault('randomize', True)
settings.setdefault('save_original_images', True)
settings.setdefault('keep_intermediate', False)
settings.setdefault('keep_original_images', False)
settings.setdefault('cmap', 'inferno')
settings.setdefault('figuresize', 10)
settings.setdefault('normalize', True)
settings.setdefault('save_dtype', 'uint16')
settings.setdefault('test_mode', False)
settings.setdefault('test_images', 10)
settings.setdefault('random_test', True)
settings.setdefault('fps', 2)
return settings
#: The fallback, and only the fallback: what a Cellpose model setting may be
#: if Cellpose itself cannot be asked. Cellpose 4 ships one stock model, so a
#: dropdown that degrades to this is still correct rather than merely
#: non-empty. The live answer comes from :func:`cellpose_model_choices`.
CELLPOSE_MODEL_CHOICES = ('cpsam',)
#: Accepted-but-mapped spellings, appended to a model dropdown so a settings
#: CSV written against Cellpose 3 still round-trips through the GUI. They all
#: resolve to ``cpsam`` — see :func:`normalize_cellpose_model_name`.
_CELLPOSE_ALIASES = ('cyto3', 'cyto2', 'nuclei')
#: Resolved at most once per process. ``None`` = not read yet.
_CELLPOSE_MODELS_CACHE = None
def _read_cellpose_models():
"""Ask Cellpose which models exist. ``()`` if it cannot be asked.
``cellpose.models.MODEL_NAMES`` is the stock list and
``cellpose.models.get_user_models()`` is the registry
``cellpose.io.add_model`` writes — a user who trains a checkpoint and
registers it should see it in the dropdown without spaCR shipping a
new release.
"""
try:
from cellpose import models as cp_models
except Exception:
LOG.debug("Cellpose could not be imported; using the shipped "
"model list", exc_info=True)
return ()
names = list(getattr(cp_models, "MODEL_NAMES", ()) or ())
try:
names += list(cp_models.get_user_models() or ())
except Exception:
LOG.debug("Could not read the Cellpose user-model registry",
exc_info=True)
out, seen = [], set()
for name in names:
name = str(name).strip()
if name and name not in seen:
seen.add(name)
out.append(name)
return tuple(out)
[docs]
def cellpose_model_choices(block=False, refresh=False):
"""Every Cellpose model on this machine, read from the Cellpose API.
The list used to be a literal, which meant spaCR could be wrong in
both directions: it offered models Cellpose 4 had removed, and it
could not offer a checkpoint the user had registered.
Importing ``cellpose.models`` costs ~2.5 s — it pulls in torch — and
this is called while a settings page is being built, so by default it
reads the API only when Cellpose is **already imported**. This is the
same bargain :func:`spacr.settings_spec._torchvision_model_names`
strikes for the torchvision zoo, and for the same measured reason. By
the time anything has segmented, Cellpose is loaded and the next
dropdown built is live.
:param block: import Cellpose if it is not loaded. For a caller that
can afford the wait and wants the definitive answer.
:param refresh: ignore the cache and ask again — for a caller that
has just registered a model.
:returns: a non-empty tuple, ``cpsam`` first. Never empty: a dropdown
with nothing in it is worse than one that is out of date.
"""
global _CELLPOSE_MODELS_CACHE
if refresh:
_CELLPOSE_MODELS_CACHE = None
if _CELLPOSE_MODELS_CACHE is not None:
return _CELLPOSE_MODELS_CACHE
if block or sys.modules.get("cellpose.models") is not None:
names = _read_cellpose_models()
if names:
_CELLPOSE_MODELS_CACHE = _cpsam_first(names)
return _CELLPOSE_MODELS_CACHE
return tuple(CELLPOSE_MODEL_CHOICES)
def _cpsam_first(names):
"""``names`` with the default model first, order otherwise preserved."""
default = CELLPOSE_MODEL_CHOICES[0]
if default in names:
return (default,) + tuple(n for n in names if n != default)
return tuple(names)
[docs]
def downloaded_zoo_models():
"""Paths of model-zoo Cellpose checkpoints already on this machine.
The live preview builds its model from the combo value. Including local zoo
checkpoints here keeps that preview aligned with a selected run model;
otherwise it would silently render stock cpsam while the user tuned
diameter and thresholds for a different checkpoint.
ONLY WHAT IS ALREADY DOWNLOADED. Listing a model that is not on disk would
put an entry in a dropdown that cannot be selected without a 1.2 GB
download starting from a combo box, which is not where anyone expects to
begin one. The picker is where downloading happens; this is where the
result of having done it shows up.
Never raises and never blocks on the network -- true again as of
2026-09-05, and it was not for a while: the ``remote=True`` below reaches
the community catalogue, which was fetched synchronously. See the comment
on that call for what it cost.
:returns: a tuple of filesystem paths, newest catalogue order.
"""
try:
import os
from . import model_zoo
out = []
for entry in model_zoo.catalogue(remote=True, include_bundled=False,
include_plugins=False, block=False):
if entry.kind != "cellpose":
continue
path = str(getattr(entry, "path", "") or "")
if path and os.path.isfile(path):
out.append(path)
return tuple(out)
except Exception: # noqa: BLE001
return ()
[docs]
def normalize_cellpose_model_name(value, object_type=None, key=None):
"""Map a stored Cellpose model setting forward onto what Cellpose 4 has.
Cellpose 4 ships exactly one stock model, ``cpsam``
(``cellpose.models.MODEL_NAMES == ['cpsam']``), and
``CellposeModel(model_type=...)`` is accepted-and-ignored. So 'cyto',
'cyto2', 'cyto3' and 'nuclei' are not four choices, they are four spellings
of cpsam — offering them in a dropdown invited users to tune a setting that
does nothing.
They are kept as accepted-but-mapped ALIASES rather than removed outright:
settings CSVs written years ago must still load. What changes is that they
are mapped here, on the way in, instead of being carried around as if they
still meant something. A path to a user-trained checkpoint is passed
through untouched — that is the one model choice that is still real.
:param value: the stored setting, e.g. 'cyto2' or '/models/my_cells.pth'.
:param object_type: 'cell'/'nucleus'/'pathogen'/'organelle' if known; used
only to make the substitution notice name the right object.
:param key: settings key the value came from, for the notice.
:returns: 'cpsam', or the checkpoint path unchanged.
"""
from .utils import LEGACY_CELLPOSE_MODELS, CPSAM_MODEL, _report_cellpose_once
if value is None:
return CPSAM_MODEL
name = str(value).strip()
if not name:
return CPSAM_MODEL
if name in LEGACY_CELLPOSE_MODELS:
where = f" ({key})" if key else ""
clause = f" for {object_type}" if object_type else ""
_report_cellpose_once(
('settings-legacy', name, object_type, key),
f"Cellpose model {name!r}{where} predates Cellpose-SAM and is no "
f"longer available; using 'cpsam'{clause}.")
return CPSAM_MODEL
return name
def _get_object_settings(object_type, settings):
"""Build per-object Cellpose/segmentation settings for cell/nucleus/pathogen.
Cellpose's ``min_size`` is the minimum of the object's ``area`` row in
``object_filters``, the row the retired ``{object}_min_area`` setting
migrates into, so undersized masks are still dropped during
segmentation.
"""
from .utils import _get_diam
from .qt.mask_engine import object_filter_area_floor
object_settings = {}
object_settings['diameter'] = _get_diam(settings['magnification'], obj=object_type)
object_settings['minimum_size'] = (object_settings['diameter']**2)/4
object_settings['maximum_size'] = (object_settings['diameter']**2)*10
object_settings['merge'] = False
object_settings['resample'] = True
object_settings['remove_border_objects'] = False
if str(settings.get('segmentation_backend') or '').strip().lower() == 'cellpose3':
object_settings['model_name'] = settings.get(f'{object_type}_model_name')
else:
object_settings['model_name'] = normalize_cellpose_model_name(
settings.get(f'{object_type}_model_name'),
object_type=object_type, key=f'{object_type}_model_name')
if object_type == 'cell':
object_settings['min_size'] = object_filter_area_floor(settings, 'cell')
object_settings['filter_size'] = False
object_settings['filter_intensity'] = False
object_settings['restore_type'] = settings.get('cell_restore_type', None)
if settings['cell_diameter'] is not None:
try:
object_settings['diameter'] = float(settings['cell_diameter'])
object_settings['minimum_size'] = (object_settings['diameter']**2)/4
object_settings['maximum_size'] = (object_settings['diameter']**2)*10
except (TypeError, ValueError):
print(f'Cell diameter must be an integer or float, got {settings["cell_diameter"]!r}')
elif object_type == 'nucleus':
object_settings['min_size'] = object_filter_area_floor(settings, 'nucleus')
object_settings['filter_size'] = False
object_settings['filter_intensity'] = False
object_settings['restore_type'] = settings.get('nucleus_restore_type', None)
if settings['nucleus_diameter'] is not None:
try:
object_settings['diameter'] = float(settings['nucleus_diameter'])
object_settings['minimum_size'] = (object_settings['diameter']**2)/4
object_settings['maximum_size'] = (object_settings['diameter']**2)*10
except (TypeError, ValueError):
print(f'Nucleus diameter must be an integer or float, got {settings["nucleus_diameter"]!r}')
elif object_type == 'pathogen':
object_settings['min_size'] = object_filter_area_floor(settings, 'pathogen')
object_settings['filter_size'] = False
object_settings['filter_intensity'] = False
object_settings['resample'] = False
object_settings['restore_type'] = settings.get('pathogen_restore_type', None)
object_settings['merge'] = settings['merge_pathogens']
if settings['pathogen_diameter'] is not None:
try:
object_settings['diameter'] = float(settings['pathogen_diameter'])
object_settings['minimum_size'] = (object_settings['diameter']**2)/4
object_settings['maximum_size'] = (object_settings['diameter']**2)*10
except (TypeError, ValueError):
print(f'Pathogen diameter must be an integer or float, got {settings["pathogen_diameter"]!r}')
else:
print(f'Object type: {object_type} not supported. Supported object types are : cell, nucleus and pathogen')
if settings['verbose']:
print(object_settings)
return object_settings
[docs]
def set_default_umap_image_settings(settings=None):
"""Return the default settings for UMAP/tSNE image-embedding plots.
:param settings: optional dict to fill in place; a new dict is created if None.
:returns: the settings dict with defaults applied.
"""
if settings is None:
settings = {}
_fold_renamed_settings(settings)
settings.setdefault('src', 'path')
settings.setdefault('row_limit', 1000)
settings.setdefault('tables', ['cell', 'cytoplasm', 'nucleus', 'pathogen'])
settings.setdefault('image_nr', 16)
settings.setdefault('dot_size', 50)
settings.setdefault('point_color', 'cluster')
settings.setdefault('point_alpha', 0.65)
settings.setdefault('outline_width', 1.0)
settings.setdefault('umap_canvas_width', 900)
settings.setdefault('umap_sidebar_width', 280)
settings.setdefault('n_neighbors', 1000)
settings.setdefault('min_dist', 0.1)
settings.setdefault('metric', 'euclidean')
settings.setdefault('tsne_perplexity', 30.0)
settings.setdefault('tsne_learning_rate', 200.0)
settings.setdefault('tsne_early_exaggeration', 12.0)
settings.setdefault('tsne_max_iter', 1000)
settings.setdefault('pca_whiten', False)
settings.setdefault('pca_svd_solver', 'auto')
settings.setdefault('isomap_n_neighbors', 15)
settings.setdefault('isomap_path_method', 'auto')
settings.setdefault('spectral_affinity', 'nearest_neighbors')
settings.setdefault('spectral_n_neighbors', 15)
settings.setdefault('random_seed', 42)
settings.setdefault('gpu', False)
settings.setdefault('eps', 0.9)
settings.setdefault('min_samples', 100)
settings.setdefault('filter_by', 'channel_0')
settings.setdefault('img_zoom', 0.5)
settings.setdefault('plot_by_cluster', True)
settings.setdefault('plot_cluster_grids', False)
settings.setdefault('remove_cluster_noise', True)
settings.setdefault('remove_highly_correlated', True)
settings.setdefault('log_data', False)
settings.setdefault('figuresize', 10)
settings.setdefault('black_background', True)
settings.setdefault('remove_image_canvas', False)
settings.setdefault('plot_outlines', True)
settings.setdefault('plot_points', True)
settings.setdefault('smooth_lines', True)
settings.setdefault('clustering', 'dbscan')
settings.setdefault('exclude', None)
settings.setdefault('col_to_compare', 'columnID')
settings.setdefault('pos', 'c1')
settings.setdefault('neg', 'c2')
settings.setdefault('mix', 'c3')
settings.setdefault('embedding_by_controls', False)
settings.setdefault('plot_images', True)
settings.setdefault('reduction_method','umap')
settings.setdefault('save_figure', False)
settings.setdefault('n_jobs', -1)
settings.setdefault('ram_guard', True)
settings.setdefault('color_by', None)
settings.setdefault('exclude_conditions', None)
settings.setdefault('exclude_rows', None)
settings.setdefault('batch_correction', 'none')
settings.setdefault('batch_column', 'plateID')
settings.setdefault('batch_control_column', None)
settings.setdefault('batch_control_values', None)
settings.setdefault('batch_covariate_column', None)
settings.setdefault('batch_combat_mean_only', False)
settings.setdefault('batch_min_samples', 3)
settings.setdefault('batch_missing_control', 'error')
settings.setdefault('analyze_clusters', False)
settings.setdefault('resnet_features', False)
settings.setdefault('verbose',True)
settings.setdefault('crop_source', 'auto')
return settings
[docs]
def get_measure_crop_settings(settings=None):
"""Return the default settings for the measure-and-crop pipeline.
Enables test mode / plotting automatically when ``test_mode`` is True.
:param settings: optional dict to fill in place; a new dict is created if None.
:returns: the settings dict with defaults applied.
"""
if settings is None:
settings = {}
_fold_renamed_settings(settings)
_set_psf_defaults(settings)
settings.setdefault('psf_measurement_source', 'original')
_set_unmix_defaults(settings)
_requested_organelle_count = organelle_count(settings)
import ast as _ast
for _k, _v in list(settings.items()):
if isinstance(_v, str) and _v.strip()[:1] in "[(":
try:
settings[_k] = _ast.literal_eval(_v)
except (ValueError, SyntaxError):
pass
settings.setdefault('src', 'path')
settings.setdefault('verbose', False)
settings.setdefault('experiment', 'experiment')
settings.setdefault('test_mode', False)
settings.setdefault('dry_run', False)
settings.setdefault('test_nr', 10)
settings.setdefault('channels', [0,1,2,3])
settings.setdefault('cloud_anonymous', False)
settings.setdefault('cloud_profile', '')
settings.setdefault('cloud_endpoint', '')
settings.setdefault('cloud_cache', '')
settings.setdefault('cloud_results', '')
settings.setdefault('save_measurements',True)
settings.setdefault('radial_dist', True)
settings.setdefault('spatial_measurements', True)
settings.setdefault('spatial_neighbor_radius', 50)
settings.setdefault('bystander_measurements', False)
settings.setdefault('bystander_reach_in_diameters', 1.0)
settings.setdefault('confluency', False)
settings.setdefault('confluency_source', 'auto')
settings.setdefault('confluency_channel', None)
settings.setdefault('confluency_window', 15)
settings.setdefault('confluency_qc_threshold', 0.8)
settings.setdefault('bleach_correction', 'none')
settings.setdefault('measure_gpu', False)
settings.setdefault('ram_guard', True)
settings.setdefault('measurement_backend', 'sqlite')
settings.setdefault('measurement_backend_target', '')
settings.setdefault('profiling', False)
settings.setdefault('profiling_metadata', '')
settings.setdefault('profiling_treatment_column', 'columnID')
settings.setdefault('profiling_negative_control', '')
settings.setdefault('profiling_normalization', 'mad_robustize')
settings.setdefault('profiling_feature_selection', [
'variance_threshold', 'frequency_threshold', 'correlation_threshold',
'drop_na_columns', 'drop_outliers'])
settings.setdefault('profiling_correlation_threshold', 0.9)
settings.setdefault('profiling_phenotype_column', '')
settings.setdefault('profiling_databases', [])
settings.setdefault('cell_cycle', False)
settings.setdefault('cell_cycle_method', 'measurements')
settings.setdefault('cell_cycle_channel', None)
settings.setdefault('cell_cycle_gates', None)
settings.setdefault('cell_cycle_mitotic_ratio', 1.8)
settings.setdefault('cell_cycle_fucci_channels', None)
settings.setdefault('cell_cycle_labels', '')
settings.setdefault('cell_cycle_model', '')
settings.setdefault('cell_cycle_epochs', 20)
settings.setdefault('wound_closure', False)
settings.setdefault('wound_source', 'texture')
settings.setdefault('wound_channel', None)
settings.setdefault('wound_window', 15)
settings.setdefault('wound_threshold', None)
settings.setdefault('wound_hours_per_frame', None)
settings.setdefault('wound_conditions', {})
settings.setdefault('intensity_calibration', False)
settings.setdefault('intensity_calibration_wells', None)
settings.setdefault('intensity_calibration_statistic', 'foreground')
settings.setdefault('intensity_calibration_offset', 0)
settings.setdefault('plate_barcode_source', '')
settings.setdefault('plate_barcodes', None)
settings.setdefault('plate_barcode_column', 'barcode')
settings.setdefault('plate_barcode_token_env', 'SPACR_LIMS_TOKEN')
settings.setdefault('time_to_event', False)
settings.setdefault('time_to_event_object', 'cell')
settings.setdefault('time_to_event_mode', 'track_end')
settings.setdefault('time_to_event_column', '')
settings.setdefault('time_to_event_threshold', None)
settings.setdefault('time_to_event_persist', 1)
settings.setdefault('time_to_event_origin', 'track')
settings.setdefault('time_to_event_min_frames', 3)
settings.setdefault('time_to_event_hours_per_frame', None)
settings.setdefault('time_to_event_group', 'well')
settings.setdefault('time_to_event_conditions', None)
settings.setdefault('time_to_event_reference', '')
settings.setdefault('time_to_event_covariates', None)
settings.setdefault('viability', False)
settings.setdefault('viability_dead_channel', None)
settings.setdefault('viability_live_channel', None)
settings.setdefault('viability_thresholds', None)
settings.setdefault('viability_negative_wells', None)
settings.setdefault('viability_positive_wells', None)
settings.setdefault('viability_plate_map', '')
settings.setdefault('cellprofiler_pipeline', '')
settings.setdefault('object_distances', True)
settings.setdefault('object_distance_maxima', True)
settings.setdefault('object_distance_intensity', True)
settings.setdefault('calculate_correlation', True)
settings.setdefault('homogeneity', True)
settings.setdefault('homogeneity_distances', [8,16,32])
settings.setdefault('voxel_size_z_um', None)
settings.setdefault('voxel_size_xy_um', None)
settings.setdefault('anisotropy', None)
settings.setdefault('save_arrays', False)
settings.setdefault('save_png',True)
settings.setdefault('use_bounding_box',False)
settings.setdefault('png_size',[224,224])
settings.setdefault('png_channel_mapping', {'r': 2, 'g': 1, 'b': 0})
settings.setdefault('normalize',False)
settings.setdefault('normalize_by','png')
settings.setdefault('crop_mode',['cell'])
settings.setdefault('dialate_pngs', False)
settings.setdefault('dialate_png_ratios', [0.2])
settings.setdefault('timelapse', False)
settings.setdefault('timelapse_objects', ['cell'])
settings.setdefault('timelapse_lineage', False)
settings.setdefault('timelapse_lineage_color_by', 'generation_time')
settings.setdefault('timelapse_lineage_max_distance', 30.0)
settings.setdefault('timelapse_lineage_min_division_h', 6.0)
settings.setdefault('plot',False)
settings.setdefault('n_jobs', _default_worker_count(reserve=2))
settings.setdefault('cell_mask_dim',4)
settings.setdefault('nucleus_mask_dim',5)
settings.setdefault('pathogen_mask_dim',6)
settings.setdefault('organelle_mask_dim',None)
settings.setdefault('cytoplasm',True)
settings.setdefault('uninfected',True)
settings.setdefault('cell_min_size',8000)
settings.setdefault('cell_max_size',None)
settings.setdefault('nucleus_max_size',None)
settings.setdefault('pathogen_max_size',None)
settings.setdefault('nucleus_min_size',2000)
settings.setdefault('pathogen_min_size',500)
settings.setdefault('organelle_min_area', 0)
settings.setdefault('organelle_type', DEFAULT_ORGANELLE_TYPE)
settings.setdefault(NUMBER_OF_ORGANELLES, _requested_organelle_count)
for _role in declared_organelle_roles(settings)[1:]:
settings.setdefault(f'{_role}_mask_dim', None)
settings.setdefault(f'{_role}_min_area', 0)
settings.setdefault(f'{_role}_type', DEFAULT_ORGANELLE_TYPE)
settings.setdefault('cytoplasm_min_size',0)
settings.setdefault('merge_edge_pathogen_cells', True)
settings.setdefault('distance_gaussian_sigma', 10)
if settings['test_mode']:
settings['verbose'] = True
settings['plot'] = True
test_imgs = settings['test_nr']
print(f'Test mode enabled with {test_imgs} images, plotting set to True')
settings.setdefault('strict_errors', None)
settings.setdefault('max_failure_rate', None)
settings.setdefault('resume', False)
settings.setdefault('summarize_organelles_by', 'cell')
if settings.get('verbose'):
explain_organelle_measurements(settings)
from .illumination import illumination_settings
for _key, _value in illumination_settings({}).items():
settings.setdefault(_key, _value)
return settings
[docs]
def set_default_classify(settings):
"""Populate defaults for the merged Classify module.
The union of Classify (CV) and Classify (ML), plus ``classifier_family``
to say which of the two a run uses. Built by CALLING both factories
rather than by copying their keys: a list here would go stale the first
time either module gained a setting, and the symptom would be a control
missing from the merged screen only.
CV is applied second and wins on the six keys the two share, because the
merged module's default family is CV.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('classifier_family', 'cv')
set_default_analyze_screen(settings)
deep_spacr_defaults(settings)
for retired in ("location_column", "positive_control_id", "negative_control_id"):
settings.pop(retired, None)
return settings
[docs]
def set_default_analyze_screen(settings):
"""Populate default settings for screen analysis (ML-based scoring).
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
_fold_renamed_settings(settings)
settings.setdefault('src', 'path')
from .training_basis import resolve_basis
if not settings.get('dataset_mode'):
settings['dataset_mode'] = resolve_basis(settings)
settings.setdefault('annotation_column', None)
settings.setdefault('model_type_ml','xgboost')
settings.setdefault('heatmap_feature','predictions')
settings.setdefault('grouping','mean')
settings.setdefault('min_max','allq')
settings.setdefault('cmap','viridis')
settings.setdefault('channel_of_interest',3)
settings.setdefault('min_cells_per_well', 25)
settings.setdefault('reg_alpha',0.1)
settings.setdefault('reg_lambda',1.0)
settings.setdefault('learning_rate',0.001)
settings.setdefault('n_estimators',1000)
settings.setdefault('test_size',0.2)
settings.setdefault('location_column','columnID')
settings.setdefault('positive_control_id','c2')
settings.setdefault('negative_control_id','c1')
settings.setdefault('exclude',None)
settings.setdefault('nuclei_limit',True)
settings.setdefault('pathogen_limit',3)
settings.setdefault('n_repeats',10)
settings.setdefault('top_features',30)
settings.setdefault('remove_low_variance_features',True)
settings.setdefault('remove_highly_correlated_features',True)
settings.setdefault('batch_correction', 'none')
settings.setdefault('batch_column', 'plateID')
settings.setdefault('batch_control_column', None)
settings.setdefault('batch_control_values', None)
settings.setdefault('batch_covariate_column', None)
settings.setdefault('batch_combat_mean_only', False)
settings.setdefault('batch_min_samples', 3)
settings.setdefault('batch_missing_control', 'error')
settings.setdefault('n_jobs',-1)
settings.setdefault('ram_guard', True)
settings.setdefault('prune_features',False)
settings.setdefault('cross_validation',True)
settings.setdefault('verbose',True)
return settings
def _set_classifier_evaluation_defaults(settings):
"""Populate shared Classify evaluation and nested-CV defaults."""
settings.setdefault('classifier_evaluation', True)
settings.setdefault('nested_cv_inner_folds', 0)
settings.setdefault('evaluation_calibration', 'temperature')
settings.setdefault('evaluation_bins', 10)
settings.setdefault('evaluation_fail_on_leakage', True)
settings.setdefault('leakage_audit_train_test', True)
settings.setdefault('leakage_hash_content', True)
settings.setdefault('leakage_require_identity', True)
return settings
#: ``old name -> new name`` for every setting that has been renamed.
#:
#: THE FOLD IS WHAT KEEPS OLD SETTINGS FILES WORKING, and it is the whole
#: reason a rename is safe to make at all: every CSV anyone has saved names
#: the old key, and a missing key is a DEFAULT rather than an error, so a
#: rename without this is a silent behaviour change on somebody else's
#: machine. `spacr.validate.RETIRED_SETTINGS` tells them what happened;
#: this makes the run work meanwhile.
#:
#: ONE ENTRY PER LANDED RENAME, added as each one lands rather than all at
#: once. 364 approved seven and each touches 16 to 29 files under `spacr/`
#: -- seven at once is one commit nobody can review and one mistake nobody
#: can bisect.
RENAMED_SETTINGS = {
"expected_end": "window_length",
"min_n": "min_observations_per_hit",
"min_cell_count": "min_cells_per_well",
"positive_control": "positive_control_id",
"negative_control": "negative_control_id",
"controls": "nontargeting_control_grnas",
"control_wells": ("stain_baseline_wells", "analysis_excluded_wells"),
"minimum_cell_count": "min_cells_per_well",
"redunction_method": "reduction_method",
"img_size": "crop_size",
"straightness_filter": "drop_straight_tracks",
"zscore_thresh": "track_outlier_zscore",
"complevel": "comp_level",
}
#: What each SEMANTIC fold does with an old value, in the words the doctor
#: uses when an old settings file names it.
#:
#: `gradient_accumulation` was a boolean beside `gradient_accumulation_steps`,
#: and `steps = 1` already IS the off state -- so the value does not move, it
#: COLLAPSES: a stored `false` means one step, whatever the step count says.
#: `_fold_gradient_accumulation` does that. Copying the boolean onto the step
#: count instead puts `False` where an `int()` is waiting.
#:
#: `Toxoplasma`, and `toxo` before it, were a boolean beside
#: `annotation_source`, and a NAME already says everything the boolean did
#: except one thing: false, which meant no annotation at all. So a stored
#: false becomes an empty `annotation_source` and a stored true becomes
#: `'toxoplasma'`, unless the file already names an organism -- the field won
#: over the boolean before the retirement, and it still does.
#: `_fold_toxoplasma` does that. A plain move would put `True` in a field
#: that expects an organism name.
SEMANTIC_FOLD_MEANINGS = {
"gradient_accumulation": (
"false means one batch per optimizer step, and true leaves "
"gradient_accumulation_steps as it is"),
"Toxoplasma": (
"true means annotation_source 'toxoplasma' and false means no "
"annotation, unless annotation_source already names an organism"),
"toxo": (
"true means annotation_source 'toxoplasma' and false means no "
"annotation, unless annotation_source already names an organism"),
}
RETIRED_OBJECT_BOUNDS = {
f"{obj}_{bound}": (obj, bound)
for obj in ("cell", "nucleus", "pathogen")
for bound in ("min_area", "max_area", "min_intensity", "max_intensity")
}
SEMANTIC_FOLD_MEANINGS.update({
key: (
f"a non-zero value becomes the "
f"{'minimum' if bound.startswith('min') else 'maximum'} of the "
f"{'area' if bound.endswith('area') else 'intensity_mean'} row "
f"for {obj} in object_filters, and 0 means no bound")
for key, (obj, bound) in RETIRED_OBJECT_BOUNDS.items()
})
#: Retired names whose migration is SEMANTIC and must not be a plain move.
#: :data:`SEMANTIC_FOLD_MEANINGS` says what each one does instead.
#:
#: DECLARED HERE RATHER THAN MERELY ABSENT FROM `RENAMED_SETTINGS`, because
#: absence is a fact with no guard and the agreement test would otherwise read
#: it as the fifth missing rename and demand it be added. This is also not
#: hypothetical: the Qt loader reads `RETIRED_SETTINGS` directly, and
#: `_translate_legacy_setting_keys({'gradient_accumulation': False})` returns
#: `{'gradient_accumulation_steps': False}` today -- the exact bug, already
#: shipped in one consumer, and the reason the run's table stays separate.
SEMANTIC_FOLDS = frozenset(SEMANTIC_FOLD_MEANINGS)
#: The two spellings the retired Toxoplasma switch was saved under, the
#: current one first. When a file carries both, the first one found wins,
#: because `toxo` became `Toxoplasma` on 2026-08-17 and a file naming both
#: was edited after that.
_TOXOPLASMA_LEGACY_KEYS = ("Toxoplasma", "toxo")
#: How many renames one key may pass through before the chain is called a
#: cycle. A key may have been renamed more than once, so a chain is normal;
#: an unbounded one is not.
_RENAME_HOP_LIMIT = 8
def _renamed_suffix_name(key):
"""One step of a ROLE-FAMILY rename, or ``None``.
Built as ``f"{role}_{new}"`` rather than through
:func:`spacr.object_roles.role_setting`, which raises for ``cytoplasm``
-- derived rather than segmented, and still declaring
``cytoplasm_min_size``.
"""
from .object_roles import RENAMED_SETTING_SUFFIXES, split_role_setting
parts = split_role_setting(key)
if parts is None:
return None
role, suffix = parts
new = RENAMED_SETTING_SUFFIXES.get(suffix)
return None if new is None else f"{role}_{new}"
def _renamed_background_switch(key):
"""The numbered name of a lettered slot background switch, or ``None``.
``remove_background_organelleb`` (2026-09-21 to 2026-09-30) is
``remove_background_organelle_2`` today, by the numbered
slot naming. A rule rather than 701 table rows, like
:func:`_renamed_suffix_name`.
:param key: the key a settings file carries.
"""
role = _legacy_background_switch_role(key)
return None if role is None else _background_switch_key(role)
def _resolve_rename(key):
"""Walk ``key`` to the end of its rename chain.
ONE WALK, used by both the resolver and the collision ordering, because
two copies of a fixed-point loop is two chances to disagree about where a
key ends up -- which is the bug this whole change exists to fix, in
miniature.
:param key: the key a settings file carries.
:returns: ``(names, hops)``. ``names`` is empty when the chain cycles or
runs past the hop limit.
"""
names, seen, hops = (str(key),), {str(key)}, 0
while hops < _RENAME_HOP_LIMIT:
step = ()
for name in names:
direct = RENAMED_SETTINGS.get(name)
if direct is None:
direct = _renamed_suffix_name(name)
if direct is None:
direct = _renamed_background_switch(name)
if direct is None:
step += (name,)
elif isinstance(direct, str):
step += (direct,)
else:
step += tuple(direct)
if step == names:
return names, hops
if any(name in seen for name in step):
return (), hops
seen.update(step)
names, hops = step, hops + 1
return (), hops
[docs]
def surviving_setting_name(key):
"""What ``key`` is called TODAY, following renames to the end.
ONE RESOLVER FOR ALL THREE CONSUMERS -- the run's fold, the doctor's
message and the Qt panel's load.
They used to answer this question three different ways. The run consulted
the rename table alone, the doctor the retirement table alone, and the Qt
panel was the only one that knew the organelle suffix rule. A generated
organelle key could therefore be migrated by the panel, ignored by the
run and unmentioned by the doctor, all at once.
A setting may have been renamed more than once. Older files need every
hop resolved so their values reach the name the current run reads.
LIVENESS IS CHECKED AT THE TERMINUS ONLY. Checking each hop would refuse
the middle of a valid chain, because an intermediate name is by
definition no longer live. Refusing when the old key is ITSELF still live
is what stops a suffix rule retiring a working setting -- seven ``_size``
keys are live and must not be touched.
:param key: the key a settings file carries.
:returns: a tuple of the names read today -- empty when ``key`` is
current, withdrawn, unknown, or resolves through a cycle.
"""
key = str(key)
if key in SEMANTIC_FOLDS:
return ()
live = expected_types
if key in live:
return ()
names, _hops = _resolve_rename(key)
if not names or names == (key,):
return ()
if not all(name in live for name in names):
return ()
return names
def _fold_renamed_settings(settings):
"""Move any renamed key onto its new name, before anything reads it.
The NEW name wins where both are present: an explicit new spelling is
a decision made later than the file it sits beside, and silently
preferring the old one would make a corrected settings file behave
like the uncorrected one.
ITERATES THE SETTINGS, NOT THE TABLE. The table is no longer a list of
old names -- six suffix rules stand in for 3,522 of them -- so there is
nothing to iterate. Reading the ~60 keys a settings file actually has is
also cheaper than the 45 table rows this used to walk.
NEAREST SPELLING WINS ON A COLLISION, by hop distance rather than by
whichever the dict happened to yield first. Two old spellings can reach
one new name: `min_cell_count` and `minimum_cell_count` both become
`min_cells_per_well`. Without an order the winner was whichever column
the CSV happened to list first.
IT SAYS WHAT IT DID. A file that read `cell_FT=0.42` as the default 100
yesterday reads it as 0.42 today, and the masks change -- correctly, but
a user who is not told will think their data changed. The common case is
not a collision, it is one old key taking effect for the first time, so
every performed migration gets a line. A current settings file has no old
keys and so prints nothing.
:param settings: the settings mapping, edited in place.
:returns: the same mapping, for chaining.
"""
if not isinstance(settings, dict):
return settings
moves = []
for old in list(settings):
if not isinstance(old, str):
continue
targets = surviving_setting_name(old)
if targets:
_names, hops = _resolve_rename(old)
moves.append((hops, old, targets))
for _hops, old, targets in sorted(moves, key=lambda row: (row[0], row[1])):
value = settings.pop(old, None)
for name in targets:
if name in settings:
LOG.warning(
"this settings file names both %r and %r. Keeping the "
"value already under %r and ignoring %r=%r.",
old, name, name, old, value)
continue
settings[name] = value
LOG.info(
"%s=%r is applied as %s. The setting was renamed and this "
"file still uses the old name.", old, value, name)
return settings
def _legacy_switch_is_on(value) -> bool:
"""Read a stored on/off value the way a settings CSV may have spelt it.
A value read back from a CSV without its type is a string, and
``bool('False')`` is True. The words a CSV writes for off are read as
off.
"""
if isinstance(value, str):
return value.strip().lower() not in (
"", "false", "0", "no", "off", "none")
return bool(value)
def _fold_toxoplasma(settings, quiet=False):
"""Let ``annotation_source`` alone say which annotation a run joins.
ONE QUESTION, ONE ANSWER. `Toxoplasma` was a boolean beside
`annotation_source`, and the name already says everything the boolean
did except one thing: false, which meant no annotation at all. So the
boolean is retired.
THE MIGRATION IS THE POINT, not the removal. A name in
`annotation_source` wins, because it won before the retirement. A blank
or missing one takes the old switch: true becomes ``'toxoplasma'`` and
false becomes the empty string, which is no annotation. Dropping the
switch instead would turn annotation ON for a file that had turned it
off.
:param settings: the settings mapping, edited in place.
:param quiet: say nothing. For a caller that folds a throwaway copy on
every read, where one line per read would be noise.
:returns: the same mapping, for chaining.
"""
if not isinstance(settings, dict):
return settings
present = [key for key in _TOXOPLASMA_LEGACY_KEYS if key in settings]
if not present:
return settings
old = present[0]
value = settings[old]
for key in present:
settings.pop(key)
named = str(settings.get('annotation_source', '') or '').strip()
if named:
if not quiet:
LOG.info(
"%s=%r was dropped: annotation_source=%r already says which "
"annotation this run joins.", old, value, named)
return settings
settings['annotation_source'] = (
'toxoplasma' if _legacy_switch_is_on(value) else '')
if not quiet:
LOG.info(
"%s=%r is applied as annotation_source=%r. %s was retired and "
"this file still uses it.", old, value,
settings['annotation_source'], old)
return settings
def _fold_object_bounds(settings, quiet=False):
"""Move the retired per-object area and intensity bounds into rows.
Mask's ``{object}_min_area``, ``_max_area``, ``_min_intensity`` and
``_max_intensity`` settings were retired on 2026-09-25 (item 511): one
filter list, ``object_filters``, holds every bound. A settings file
written before carries them, and its values must still apply, so they
become rows of ``object_filters`` for that object through
:func:`spacr.qt.mask_engine.legacy_filters`, where 0 was off and stays
off.
A row the file already has for the same property wins: a filter list is
the later spelling, written by someone who has seen the new form.
A value that is not a finite number of zero or more is LEFT where it is,
unmigrated: the old bound was refused by the run, the doctor and the
Live preview (``spacr.utils._validated_intensity_bounds``), and moving
it would turn that refusal into a silent "off".
:param settings: the settings mapping, edited in place.
:param quiet: say nothing, for a caller that folds a throwaway copy.
:returns: the same mapping, for chaining.
"""
if not isinstance(settings, dict):
return settings
present = [key for key in RETIRED_OBJECT_BOUNDS if key in settings]
if not present:
return settings
from .qt.mask_engine import legacy_filters, parse_object_filters
import math
by_object = {}
for key in present:
obj, bound = RETIRED_OBJECT_BOUNDS[key]
value = settings[key]
try:
number = float(value) if value not in (None, '') else 0.0
except (TypeError, ValueError):
number = float('nan')
if not math.isfinite(number) or number < 0:
if not quiet:
LOG.warning(
"%s=%r is not a bound of zero or more, so it was left "
"where it is rather than moved into object_filters.",
key, value)
continue
settings.pop(key)
by_object.setdefault(obj, {})[bound] = number
if not by_object:
return settings
table = parse_object_filters(settings.get('object_filters'))
for obj, bounds in by_object.items():
rows = legacy_filters(**bounds)
if not rows:
continue
kept = list(table.get(obj) or [])
named = {str(row.get('property')) for row in kept
if isinstance(row, dict)}
added = [row for row in rows if row['property'] not in named]
if not added:
continue
table[obj] = kept + added
if not quiet:
LOG.info(
"The retired %s bounds %r are applied as object_filters "
"rows %r.", obj, bounds, added)
settings['object_filters'] = table
return settings
def _fold_gradient_accumulation(settings):
"""Let ``gradient_accumulation_steps`` alone say whether to accumulate.
ONE QUESTION, ONE ANSWER. `gradient_accumulation` was a boolean sitting
beside the step count, and ``steps = 1`` already IS the off state -- one
batch per optimizer step is ordinary training. Two keys for one decision
is a pair that can disagree: on with one step, off with eight, and
nothing saying which wins.
THE MIGRATION IS THE POINT, not the removal. Retiring the flag on its own
would silently turn accumulation ON for everyone who had written
``gradient_accumulation: false`` beside the default four steps -- the
value would be ignored, the step count would stand, and their training
would change without a word. So a stored ``false`` is honoured by
collapsing the step count to 1, which says the same thing in the new
spelling. A stored ``true`` needs nothing; the steps already say how many.
:param settings: the settings mapping, edited in place.
:returns: the same mapping, for chaining.
"""
if 'gradient_accumulation' not in settings:
return settings
legacy = settings.pop('gradient_accumulation')
if legacy is False or str(legacy).strip().lower() in ('false', '0', 'no'):
settings['gradient_accumulation_steps'] = 1
return settings
[docs]
def set_default_train_test_model(settings):
"""Populate default settings for the train/test classifier training pipeline.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.pop('custom_model', None)
cores = _default_worker_count(reserve=2)
settings.setdefault('src','path')
settings.setdefault('train',True)
settings.setdefault('test',False)
settings.setdefault('classes', {})
settings.setdefault('class_folder_names', ['nc','pc'])
settings.setdefault('model_type','maxvit_t')
settings.setdefault('optimizer_type','adamw')
settings.setdefault('schedule','cosine')
settings.setdefault('loss_type','focal_loss')
settings.setdefault('normalize',True)
settings.setdefault('image_size',224)
settings.setdefault('batch_size',64)
settings.setdefault('epochs',100)
settings.setdefault('plot',True)
settings.setdefault('tensorboard',True)
settings.setdefault('val_split',0.1)
settings.setdefault('learning_rate',0.001)
settings.setdefault('weight_decay',0.00001)
settings.setdefault('dropout_rate',0.1)
settings.setdefault('init_weights',True)
settings.setdefault('amsgrad',True)
settings.setdefault('use_checkpoint',True)
settings.setdefault('mixed_precision', False)
settings.setdefault('gradient_accumulation_steps',4)
_fold_gradient_accumulation(settings)
settings.setdefault('intermedeate_save',True)
settings.setdefault('resume_checkpoint','')
settings.setdefault('custom_model_path','')
settings.setdefault('pin_memory',False)
settings.setdefault('n_jobs',cores)
settings.setdefault('ram_guard', True)
settings.setdefault('train_channels',['r','g','b'])
settings.setdefault('augment',False)
settings.setdefault('verbose',False)
settings.setdefault('class_balance','none')
settings.setdefault('cross_validation_folds',0)
settings.setdefault('cross_validation_enabled',False)
settings.setdefault('cv_group_by','well')
settings.setdefault('holdout_plate', None)
_set_classifier_evaluation_defaults(settings)
settings.setdefault('strict_errors', None)
settings.setdefault('max_failure_rate', None)
settings.setdefault('crop_source', 'auto')
return settings
[docs]
def set_generate_training_dataset_defaults(settings):
"""Populate default settings for generating a labeled training dataset.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src','path')
settings.setdefault('tables', ['cell', 'nucleus', 'pathogen', 'cytoplasm'])
settings.setdefault('dataset_mode','metadata')
settings.setdefault('annotation_column','test')
settings.setdefault('metadata_item_1_name',None)
settings.setdefault('metadata_item_1_value',None)
settings.setdefault('metadata_item_2_name',None)
settings.setdefault('metadata_item_2_value',None)
settings.setdefault('test_split',0.1)
settings.setdefault('cv_group_by','well')
settings.setdefault('holdout_plate', None)
settings.setdefault('class_metadata',[['c1'],['c2']])
_fold_the_classes(settings)
settings.setdefault('channel_of_interest',3)
settings.setdefault('nuclei_limit',True)
settings.setdefault('pathogen_limit',True)
settings.setdefault('random_seed',42)
return settings
[docs]
def deep_spacr_defaults(settings):
"""Populate default settings for the end-to-end deep_spacr training pipeline.
Covers dataset generation, model training/testing and applying the trained
model to the dataset in a single settings dict.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.pop('custom_model', None)
if settings.get('extract_channels') and not settings.get(
'train_channels'):
settings['train_channels'] = list(settings['extract_channels'])
settings.pop('extract_channels', None)
cores = _default_worker_count(reserve=4)
settings.setdefault('src','path')
settings.setdefault('dataset_mode','metadata')
settings.setdefault('annotation_column','test')
settings.setdefault('classes', {})
settings.setdefault('class_folder_names', ['nc','pc'])
settings.setdefault('test_split',0.1)
settings.setdefault('class_metadata',[['c1'],['c2']])
settings.setdefault('channel_of_interest',3)
settings.setdefault('tables', ['cell', 'nucleus', 'pathogen',
'cytoplasm'])
settings.setdefault('custom_model_path','')
settings.setdefault('train',True)
settings.setdefault('test',False)
settings.setdefault('model_type','maxvit_t')
settings.setdefault('optimizer_type','adamw')
settings.setdefault('schedule','cosine')
settings.setdefault('loss_type','auto')
settings.setdefault('normalize',True)
settings.setdefault('image_size',224)
settings.setdefault('batch_size',64)
settings.setdefault('epochs',100)
settings.setdefault('plot',True)
settings.setdefault('tensorboard',True)
settings.setdefault('val_split',0.1)
settings.setdefault('learning_rate',0.001)
settings.setdefault('weight_decay',0.00001)
settings.setdefault('dropout_rate',0.1)
settings.setdefault('init_weights',True)
settings.setdefault('amsgrad',True)
settings.setdefault('use_checkpoint',True)
settings.setdefault('mixed_precision', False)
settings.setdefault('gradient_accumulation_steps',4)
_fold_gradient_accumulation(settings)
settings.setdefault('label_smoothing',0.1)
settings.setdefault('focal_gamma',2.0)
settings.setdefault('focal_alpha',None)
settings.setdefault('logit_adjust_tau',1.0)
settings.setdefault('early_stopping_patience',0)
settings.setdefault('intermedeate_save',True)
settings.setdefault('resume_checkpoint','')
settings.setdefault('pin_memory',False)
settings.setdefault('n_jobs',cores)
settings.setdefault('ram_guard', True)
settings.setdefault('train_channels',['r','g','b'])
settings.setdefault('augment',False)
settings.setdefault('verbose',True)
settings.setdefault('apply_model_to_dataset',True)
settings.setdefault('file_metadata',None)
settings.setdefault('sample',None)
settings.setdefault('experiment','experiment')
settings.setdefault('score_threshold',0.5)
from .inference_augmentation import DEFAULTS as inference_defaults
for key, value in inference_defaults.items():
settings.setdefault(key, value)
settings.setdefault('dataset','')
settings.setdefault('model_path','')
settings.setdefault('file_type','cell_png')
settings.setdefault('generate_training_dataset', True)
settings.setdefault('balance_to_smallest', True)
settings.setdefault('generate_full_dataset', False)
settings.setdefault('tar_path','')
settings.setdefault('n_top_examples',20)
settings.setdefault('random_seed',42)
settings.setdefault('image_source', settings.get('crop_source')
or 'load_images')
settings['image_source'] = _canonical_image_source(
settings['image_source'])
settings['crop_source'] = settings['image_source']
settings.setdefault('object_array', 'cell')
settings['coordinate_columns'] = _coordinate_columns_for(
settings.get('object_array'))
settings.setdefault('stream_method', 'column')
settings.setdefault('channel_arrays', [0, 1, 2])
settings.setdefault('bounding_box', True)
settings.setdefault('crop_shape', 'bounding_box')
settings.setdefault('strict_errors',None)
settings.setdefault('max_failure_rate',None)
settings.setdefault('class_balance','none')
settings.setdefault('cross_validation_folds',0)
settings.setdefault('cross_validation_enabled',False)
settings.setdefault('cv_group_by','well')
settings.setdefault('holdout_plate', None)
_set_classifier_evaluation_defaults(settings)
return settings
[docs]
def get_train_test_model_settings(settings):
"""Populate default settings for the train/test classifier settings dict.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.pop('custom_model', None)
settings.setdefault('src', 'path')
settings.setdefault('train', True)
settings.setdefault('test', False)
settings.setdefault('classes', {})
settings.setdefault('class_folder_names', ['nc','pc'])
settings.setdefault('train_channels', ['r','g','b'])
settings.setdefault('model_type', 'maxvit_t')
settings.setdefault('optimizer_type', 'adamw')
settings.setdefault('schedule', 'cosine')
settings.setdefault('loss_type', 'focal_loss')
settings.setdefault('normalize', True)
settings.setdefault('image_size', 224)
settings.setdefault('batch_size', 64)
settings.setdefault('epochs', 100)
settings.setdefault('plot', True)
settings.setdefault('tensorboard', True)
settings.setdefault('val_split', 0.1)
settings.setdefault('learning_rate', 0.0001)
settings.setdefault('weight_decay', 0.00001)
settings.setdefault('dropout_rate', 0.1)
settings.setdefault('init_weights', True)
settings.setdefault('amsgrad', True)
settings.setdefault('use_checkpoint', True)
settings.setdefault('mixed_precision', False)
settings.setdefault('gradient_accumulation_steps', 4)
_fold_gradient_accumulation(settings)
settings.setdefault('intermedeate_save',True)
settings.setdefault('resume_checkpoint','')
settings.setdefault('custom_model_path','')
settings.setdefault('pin_memory', True)
settings.setdefault('n_jobs', 30)
settings.setdefault('ram_guard', True)
settings.setdefault('augment', True)
settings.setdefault('verbose', True)
settings.setdefault('label_smoothing', 0.1)
settings.setdefault('focal_gamma', 2.0)
settings.setdefault('focal_alpha', None)
settings.setdefault('logit_adjust_tau', 1.0)
settings.setdefault('early_stopping_patience', 0)
settings.setdefault('class_balance', 'none')
settings.setdefault('cross_validation_folds', 0)
settings.setdefault('cross_validation_enabled', False)
settings.setdefault('cv_group_by', 'well')
_set_classifier_evaluation_defaults(settings)
return settings
[docs]
def get_analyze_recruitment_default_settings(settings):
"""Populate default settings for the recruitment-analysis pipeline.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src', 'path')
settings.setdefault('target','protein')
settings.setdefault('cell_types',['HeLa'])
settings.setdefault('cell_plate_metadata',None)
settings.setdefault('pathogen_types',['pathogen_1', 'pathogen_2'])
settings.setdefault('pathogen_plate_metadata',[['c1', 'c2', 'c3'],['c4','c5', 'c6']])
settings.setdefault('treatments',['cm', 'lovastatin'])
settings.setdefault('treatment_plate_metadata',[['r1', 'r2','r3'], ['r4', 'r5','r6']])
settings.setdefault('channel_dims',[0,1,2,3])
settings.setdefault('cell_chann_dim',3)
settings.setdefault('nucleus_chann_dim',0)
settings.setdefault('pathogen_chann_dim',2)
settings.setdefault('channel_of_interest',2)
settings.setdefault('plot',True)
settings.setdefault('plot_nr',3)
settings.setdefault('plot_control',True)
settings.setdefault('figuresize',10)
settings.setdefault('pathogen_limit',10)
settings.setdefault('nuclei_limit',1)
settings.setdefault('cells_per_well',0)
settings.setdefault('pathogen_size_range',[0,100000])
settings.setdefault('nucleus_size_range',[0,100000])
settings.setdefault('cell_size_range',[0,100000])
settings.setdefault('pathogen_intensity_range',[0,100000])
settings.setdefault('nucleus_intensity_range',[0,100000])
settings.setdefault('cell_intensity_range',None)
settings.setdefault('target_intensity_min',1)
return settings
[docs]
def get_default_test_cellpose_model_settings(settings):
"""Populate default settings for testing a Cellpose model on a dataset.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src','path')
settings.setdefault('model_path','path')
settings.setdefault('save',True)
settings.setdefault('normalize',True)
settings.setdefault('percentiles',(2,98))
settings.setdefault('batch_size',50)
settings.setdefault('CP_probability',0)
settings.setdefault('FT',0.4)
settings.setdefault('target_size',1000)
return settings
[docs]
def get_default_apply_cellpose_model_settings(settings):
"""Populate default settings for applying a Cellpose model to a dataset.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src','path')
settings.setdefault('model_path','path')
settings.setdefault('save',True)
settings.setdefault('normalize',True)
settings.setdefault('percentiles',(2,98))
settings.setdefault('batch_size',50)
settings.setdefault('CP_probability',0)
settings.setdefault('FT',0.4)
settings.setdefault('circularize',False)
settings.setdefault('target_size',1000)
return settings
[docs]
def default_settings_analyze_percent_positive(settings):
"""Populate default settings for the "percent positive" per-well analysis.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src','path')
settings.setdefault('tables',['cell'])
settings.setdefault('filter_1',['cell_area',1000])
settings.setdefault('value_col','cell_channel_2_mean_intensity')
settings.setdefault('threshold',2000)
return settings
[docs]
def get_analyze_reads_default_settings(settings):
"""Populate default settings for analyzing FASTQ read barcodes.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src', 'path')
settings.setdefault('chunk_size', 1000000)
settings.setdefault('test', False)
return settings
[docs]
def get_map_barcodes_default_settings(settings):
"""Populate default settings for mapping barcodes to gRNAs and plates.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src', 'path')
settings.setdefault('test', False)
settings.setdefault('verbose', True)
return settings
[docs]
def get_train_cellpose_default_settings(settings):
"""Populate default settings for training a Cellpose model.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
defaults = dict(
src='', mask_src='', test_src='', test_mask_src='', save_path='',
model_name='new_model', base_model='cpsam', learning_rate=1e-5,
weight_decay=0.1, batch_size=1, n_epochs=100,
normalize=True, percentiles=[1, 99], channels=None, channel_axis=None,
min_train_masks=5, max_train_images=None, nimg_per_epoch=None,
nimg_test_per_epoch=None, scale_range=0.5, save_every=100,
save_each=False,
)
for key, value in defaults.items():
settings.setdefault(key, value)
return settings
[docs]
def set_generate_dataset_defaults(settings):
"""Populate default settings for the generic dataset-generation step.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src','path')
settings.setdefault('file_metadata',None)
settings.setdefault('experiment','experiment_1')
settings.setdefault('sample',None)
settings.setdefault('crop_source', 'auto')
return settings
#: ``inference`` -> the ``analysis_mode`` it selects. 'auto' is decided from
#: the design at run time by :func:`spacr.ml.resolve_auto_inference`, not here,
#: because the number of guides and wells is not known until the CSVs are read.
#: The families whose `alpha` is a real penalty rather than a knob they
#: ignore. Named once: the GUI, the defaults and the refusal all have to agree
#: about which families this is, and three copies of a tuple is how they stop.
PENALISED_REGRESSION_TYPES = ('ridge', 'lasso', 'elasticnet')
INFERENCE_MODES = {
'auto': None,
'parametric': 'regression',
'nonparametric': 'guide_permutation',
'permutation': 'guide_permutation',
'regression': 'regression',
'guide_permutation': 'guide_permutation',
}
#: ``analysis_unit`` -> whether scores are collapsed per well before fitting.
ANALYSIS_UNITS = ('well', 'cell')
#: Values for selecting the granularity of a fixed-effects regression.
#:
#: ``'both'`` runs separate gRNA- and gene-level fits. It does not put both
#: terms into one formula: ``gene_fraction`` is the sum of the constituent
#: gRNA fractions, so the combined design would be collinear and its
#: coefficients would not be independently identifiable. ``'mixed'`` instead
#: models both levels by nesting each guide within its gene, so
#: :func:`get_setting_dependencies` disables this setting for mixed models.
REGRESSION_LEVELS = ('both', 'grna', 'gene')
def _resolve_regression_backend(value):
"""Normalize a submitted regression backend to its display label.
Display labels retain the ``(CPU)`` or ``(GPU)`` suffix so saved settings
record their hardware requirement and both front ends can select the same
value they display. Short names and supported aliases remain valid input.
:param value: display label, short backend name, alias, or ``None``.
:returns: a label from ``regression_spec.REGRESSION_BACKENDS``.
:raises ValueError: if ``value`` does not identify a supported backend.
"""
from .regression_backends import backend_label
return backend_label(value)
def _resolve_regression_analysis_choices(settings):
"""Map ``inference`` / ``analysis_unit`` / ``regression_type='auto'``.
These three are the readable spellings of decisions spaCR already made,
and this is the single place they are translated:
* ``inference`` selects ``analysis_mode``. ``'auto'`` is left for
:func:`spacr.ml.resolve_auto_inference`, which can count the guides and
wells that this function cannot see.
* ``analysis_unit='cell'`` sets ``agg_type=None``, which is how spaCR has
always meant "fit per object instead of per well". Spelling it out stops
a user clearing a dropdown and silently changing the unit of analysis.
* ``regression_type='auto'`` becomes ``None``, the historical value that
makes :func:`spacr.ml.regression` pick the family from the response.
An explicit ``analysis_mode`` in the same dict wins over a non-auto
``inference`` only when ``inference`` was left at its default, so a CSV
written by an older spaCR keeps its meaning.
"""
inference = str(settings.get('inference', 'auto')).strip().lower()
if inference not in INFERENCE_MODES:
raise ValueError(
f"inference={settings.get('inference')!r} is not one of "
f"{sorted(set(INFERENCE_MODES))}. 'parametric' fits the "
f"simultaneous model, 'nonparametric' runs the "
f"permutation test, and 'auto' picks whichever the design can "
f"support.")
settings['inference'] = inference
selected = INFERENCE_MODES[inference]
if selected is not None:
settings['analysis_mode'] = selected
unit = str(settings.get('analysis_unit', 'well')).strip().lower()
if unit not in ANALYSIS_UNITS:
raise ValueError(
f"analysis_unit={settings.get('analysis_unit')!r} must be 'well' "
f"or 'cell'. 'well' collapses each well's objects with agg_type "
f"first; 'cell' regresses the individual objects.")
settings['analysis_unit'] = unit
if unit == 'cell':
settings['agg_type'] = None
chosen = str(settings.get('inference', 'auto')).strip().lower()
if chosen in ('', 'auto'):
settings['inference'] = 'parametric'
settings['analysis_mode'] = 'regression'
elif chosen in ('nonparametric', 'permutation'):
raise ValueError(
"analysis_unit='cell' and inference='nonparametric' cannot "
"both hold: the permutation test compares each guide across "
"WELLS and needs one row per well, and 'cell' gives one row "
"per object. Set analysis_unit='well' with an agg_type such "
"as 'mean' to keep the permutation test, or "
"inference='parametric' to keep per-object fitting.")
elif settings.get('agg_type') is None and unit == 'well':
settings['agg_type'] = 'mean'
level = str(settings.get('level', 'both')).strip().lower()
if level not in REGRESSION_LEVELS:
raise ValueError(
f"level={settings.get('level')!r} must be one of "
f"{list(REGRESSION_LEVELS)}. 'grna' fits one coefficient per "
f"guide, 'gene' one per gene from the summed guide fraction, and "
f"'both' fits those two models SEPARATELY and BH-corrects each "
f"within itself. A value spaCR does not recognise is refused "
f"rather than falling back to 'both', because falling back would "
f"quietly fit two models when one was asked for.")
settings['level'] = level
if isinstance(settings.get('regression_type'), str) and \
settings['regression_type'].strip().lower() == 'auto':
settings['regression_type'] = None
return settings
def _reject_a_threshold_that_cannot_mean_what_it_says(settings):
"""Refuse the hit-calling numbers that are off by a factor of a hundred.
Every value checked here has a plausible-looking spelling that is wrong
by exactly that much -- ``p_threshold_alpha=5`` for "5%",
``rra_alpha=25`` for "the top 25%" -- and none of them CRASHES anything
downstream. An alpha of 5 calls every coefficient a hit and writes the
list out; an rra_alpha of 25 scores the whole ranking as the top of the
ranking. Caught here the mistake costs a sentence, and the sentence names
the legal range; caught later it costs the run that produced the hits,
because nothing about the output says the threshold was impossible.
``p_threshold_kind`` is a closed pair rather than a range: 'Adjusted' or
'bh' falling back to 'adjusted' would silently answer a question the user
asked in words spaCR does not use.
:param settings: the resolved regression settings, read in place.
:raises ValueError: on any value outside its documented range.
"""
kind = settings['p_threshold_kind']
if not isinstance(kind, str) or kind.lower() not in ('adjusted', 'raw'):
raise ValueError(
f"p_threshold_kind must be 'adjusted' or 'raw'; got {kind!r}. "
f"'adjusted' cuts on the multiple-testing-corrected P value "
f"produced by multiple_testing_method, 'raw' on the "
f"per-coefficient P value the fit reports.")
def _number(key, low, high, *, low_open=True, high_open=True):
"""Read one setting as a number in range, or raise saying which.
A bool is REFUSED even though it is an int in Python: ``True`` as a
threshold is a value nobody typed on purpose, and accepting it as 1
silently changes what the run does.
"""
value = settings[key]
if isinstance(value, bool) or not isinstance(value, (int, float)):
raise ValueError(f"{key} must be a number; got {value!r}.")
below = value <= low if low_open else value < low
above = value >= high if high_open else value > high
if below or above:
raise ValueError(
f"{key} must be between {low} and {high}"
f"{' exclusive' if low_open and high_open else ''}; got "
f"{value!r}. A probability is a fraction: 0.05, not 5.")
_number('p_threshold_alpha', 0, 1)
_number('rra_alpha', 0, 1, high_open=False)
permutations = settings['rra_permutations']
if isinstance(permutations, bool) or not isinstance(permutations, int) \
or permutations < 1:
raise ValueError(
f"rra_permutations must be a positive integer; got "
f"{permutations!r}. The smallest P value the null can report is "
f"about 1/rra_permutations.")
penalty = settings['group_lasso_lambda']
unanswered = penalty is None or (
isinstance(penalty, str) and penalty.strip().lower() in ('', 'auto'))
if unanswered:
settings['group_lasso_lambda'] = 'auto'
chose_the_penalty = not unanswered
if chose_the_penalty and (
isinstance(penalty, bool)
or not isinstance(penalty, (int, float)) or penalty < 0):
raise ValueError(
f"group_lasso_lambda must be zero or positive, or 'auto' to "
f"cross-validate it; got {penalty!r}. A negative penalty rewards "
f"large coefficients instead of shrinking them.")
for key in ('count_grna_column', 'count_value_column'):
name = settings[key]
if not isinstance(name, str) or not name.strip():
raise ValueError(
f"{key} must name a column of the count CSV; got {name!r}. "
f"An empty name cannot be looked up, and the failure would "
f"come out of pandas rather than out of the settings.")
for key, choices in (
('independent_variable_layout', {'auto', 'long', 'wide'}),
('model_data_layout', {'long', 'wide'}),
):
value = str(settings[key] or '').strip().lower()
if value not in choices:
raise ValueError(
f"{key}={settings[key]!r}; choose one of {sorted(choices)}."
)
settings[key] = value
return settings
#: What the panel posted for `group_lasso_lambda` before it defaulted to
#: 'auto'. Every settings file written up to 2026-08-22 carries it, and it
#: means "untouched" in all of them.
LEGACY_GROUP_LASSO_LAMBDA = 0.05
[docs]
def get_check_cellpose_models_default_settings(settings):
"""Populate default settings for the "check Cellpose models" utility.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('batch_size', 10)
settings.setdefault('CP_prob', 0)
settings.setdefault('flow_threshold', 0.4)
settings.setdefault('save', True)
settings.setdefault('normalize', True)
settings.setdefault('channels', [0,0])
settings.setdefault('percentiles', None)
settings.setdefault('invert', False)
settings.setdefault('plot', True)
settings.setdefault('diameter', 40)
settings.setdefault('grayscale', True)
settings.setdefault('remove_background', False)
settings.setdefault('background', 100)
settings.setdefault('Signal_to_noise', 5)
settings.setdefault('verbose', False)
settings.setdefault('resize', False)
settings.setdefault('target_height', None)
settings.setdefault('target_width', None)
return settings
[docs]
def get_identify_masks_finetune_default_settings(settings):
"""Populate default settings for fine-tuning mask identification.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src', 'path')
settings.setdefault('model_name', 'cpsam')
settings.setdefault('custom_model', None)
settings.setdefault('channels', [0,0])
settings.setdefault('background', 100)
settings.setdefault('remove_background', False)
settings.setdefault('Signal_to_noise', 10)
settings.setdefault('CP_prob', 0)
settings.setdefault('diameter', 30)
settings.setdefault('batch_size', 50)
settings.setdefault('flow_threshold', 0.4)
settings.setdefault('save', False)
settings.setdefault('verbose', False)
settings.setdefault('normalize', True)
settings.setdefault('percentiles', None)
settings.setdefault('invert', False)
settings.setdefault('resize', False)
settings.setdefault('target_height', None)
settings.setdefault('target_width', None)
settings.setdefault('rescale', False)
settings.setdefault('resample', False)
settings.setdefault('grayscale', True)
settings.setdefault('fill_in', True)
return settings
q = None
#: Module blurbs, keyed by app key.
#:
#: RESTORED. This table sat between two widget builders from the previous
#: interface and was deleted along with them, which broke registration
#: itself -- `register_defaults` reads it whenever a module registers with
#: a description. Kept here, with nothing above or below it that could
#: take it down again by association.
descriptions = {
'mask': (
"Generate labeled masks for cells, nuclei, pathogens and organelles "
"from microscopy images. Configure timelapse or motility acquisition "
"settings when applicable; the outputs are consumed by Measure."
),
'measure': (
"Measure morphology, intensity and spatial relationships for labeled "
"objects, and optionally export per-object image crops. Requires mask "
"arrays produced by Make Masks or an equivalent compatible source."
),
'classify': (
"Train, validate or apply a PyTorch image classifier to measured "
"object crops. Keep training, validation and test groups independent "
"at the selected experimental unit."
),
'umap': (
"Compute UMAP, t-SNE or PCA embeddings from measured features and "
"optionally display linked object images. Requires a measurement "
"database; image overlays additionally require compatible crops."
),
'train_cellpose': (
"Fine-tune a Cellpose model from paired microscopy images and label "
"masks. This workflow is available from Make Masks and records the "
"training configuration and model artifact."
),
'ml_analyze': (
"Fit and evaluate feature-based classifiers using measured object "
"properties and grouped validation. This workflow is available from "
"Classify and reports held-out performance and feature importance."
),
'cellpose_masks': (
"Apply a Cellpose model to microscopy images and write compatible "
"label masks. This workflow is available from Make Masks; verify mask "
"quality before downstream measurement."
),
'cellpose_all': (
"Segment microscopy images with Cellpose and extract object "
"measurements in sequence. Review the generated masks before "
"interpreting the measurement tables."
),
'map_barcodes': (
"Extract plate and guide barcodes from paired sequencing reads, map "
"them against reference CSV files and report unmatched reads. Verify "
"the regex, barcode orientation and reference columns before a full run."
),
'regression': (
"Fit grouped regression models that relate image-derived phenotypes "
"to guide or gene abundance. Define controls, exclusions, aggregation "
"and the independent experimental unit before fitting."
),
'activation': (
"Generate attribution maps for a trained image classifier using "
"methods such as Grad-CAM, SmoothGrad, occlusion or integrated "
"gradients. These maps describe model sensitivity and do not by "
"themselves establish a biological mechanism."
),
'analyze_plaques': (
"Segment plaque images and quantify plaque area, intensity and shape "
"across fields or conditions. Inspect segmentation quality before "
"comparing summary measurements."
),
'recruitment': (
"Quantify signal recruitment to measured host-cell, nucleus or "
"pathogen compartments and compare predefined experimental groups. "
"Requires compatible measurement tables and plate metadata."
),
}
expected_types = {
"psf_measurement_source": str,
"psf_operation": str, "psf_source": str, "psf_objective": str,
"psf_path": (str, type(None)),
"psf_image_sampling_um": (list, type(None)),
"psf_kernel_sampling_um": (list, type(None)),
"psf_fwhm_um": (list, type(None)), "psf_iterations": int,
"unmix": bool, "unmix_controls": str,
"unmix_background_percentile": (float, int),
"n2v_denoise": bool, "n2v_model": str, "n2v_epochs": int,
"enhance_background": str, "enhance_background_radius": int,
"enhance_background_scale": float,
"enhance_denoise": str, "enhance_denoise_strength": float,
"enhance_percentile_clip": bool, "enhance_percentile_low": float,
"enhance_percentile_high": float,
"enhance_gamma": float, "enhance_log": bool, "enhance_log_gain": float,
"enhance_sqrt": bool,
"enhance_clahe": bool, "enhance_clahe_tile": int, "enhance_clahe_clip": float,
"enhance_equalize": bool,
"enhance_sharpen": bool, "enhance_sharpen_radius": float,
"enhance_sharpen_amount": float,
"src": (str, list),
"illumination_correction": bool,
"illumination_per_plate": bool,
"illumination_qc": bool,
"illumination_dark": float,
"illumination_degree": int,
"illumination_max_fields": int,
"illumination_estimator": str,
"illumination_model": str,
"illumination_vendor_profile": str,
"illumination_vendor_channel_map": str,
"illumination_on_missing": str,
"dst": str,
"db_path": str,
"predictions_file": str,
"path_column": str,
"metadata_type": str,
"custom_regex": (str, type(None)),
"cov_type": (str, type(None)),
"experiment": str,
"channels": list,
"magnification": int,
"plaque_model": str,
"plaque_mode": str,
"figure_detector": str,
"figure_imgsz": str,
"figure_confidence": float,
"figure_read_text": bool,
"confirm_annotations": bool,
"text_reach_above": float,
"text_reach_left": float,
"text_reach_below": float,
"text_use_above": bool,
"text_use_left": bool,
"text_use_below": bool,
"text_panel_reach": float,
"text_min_confidence": float,
"text_ignore": str,
"text_order": str,
"text_separator": str,
"text_reread": bool,
"text_reread_scale": int,
"well_detection": (str, bool),
"well_confidence": float,
"well_pad": int,
"plate_format": (str, type(None)),
"well_diameter_mm": (float, int, type(None)),
"plaque_estimate_growth": bool,
"plaque_growth_reference_um": (float, int),
"plaque_growth_reference_hours": (float, int),
"plaque_pixels_per_um": (float, int, type(None)),
"plaque_formation_hours": (float, int, type(None)),
"colony_counting": bool,
"colony_dilution": (float, int, dict, str),
"colony_plated_volume_ul": (float, int),
"colony_too_many": (int, float, type(None)),
"colony_too_few": (int, float, type(None)),
"colony_polarity": str,
"colony_threshold": (float, int),
"colony_min_area_px": (float, int, type(None)),
"colony_detector": (str, type(None)),
"nucleus_channel": (int, type(None)),
"nucleus_background": int,
"nucleus_signal_to_noise": float,
"nucleus_cellprob_threshold": float,
"nucleus_flow_threshold": (int, float),
"cell_channel": (int, type(None)),
"cell_background": (int, float),
"cell_signal_to_noise": (int, float),
"cell_cellprob_threshold": (int, float),
"cell_flow_threshold": (int, float),
"pathogen_channel": (int, type(None)),
"pathogen_background": (int, float),
"pathogen_signal_to_noise": (int, float),
"pathogen_cellprob_threshold": (int, float),
"pathogen_flow_threshold": (int, float),
"preprocess": bool,
"masks": bool,
"examples_to_plot": int,
"randomize": bool,
"timelapse": bool,
"timelapse_displacement": int,
"timelapse_memory": int,
"timelapse_frame_limits": (list, type(None)),
"timelapse_remove_transient": bool,
"timelapse_mode": str,
"trackastra_model": str,
"trackastra_linking": str,
"ultrack_max_distance": float,
"ultrack_division_weight": float,
"ultrack_contour_sigma": float,
"ultrack_n_workers": int,
"timeflows_model": (str, type(None)),
"timelapse_objects": list,
"timelapse_lineage": bool,
"timelapse_lineage_color_by": str,
"timelapse_lineage_max_distance": (int, float),
"timelapse_lineage_min_division_h": (int, float, type(None)),
"timelapse_events": bool,
"timelapse_events_annotations": (str, type(None)),
"timelapse_events_model": (str, type(None)),
"timelapse_events_window": int,
"timelapse_events_threshold": (int, float),
"timelapse_events_conditions": (list, type(None)),
"timelapse_events_encoder": str,
"timelapse_events_video_checkpoint": (str, type(None)),
"timelapse_events_video_channels": (list, type(None)),
"timelapse_events_video_device": str,
"fps": int,
"lower_percentile": (int, float),
"merge_pathogens": bool,
"z_stack": bool,
"z_segmentation_mode": str,
"z_axis": (int, type(None)),
"z_projection": (str, type(None)),
"anisotropy": (int, float, type(None)),
"voxel_size_z_um": (int, float, type(None)),
"voxel_size_xy_um": (int, float, type(None)),
"stitch_threshold": (int, float),
"t_stack": bool,
"t_axis_order": (str, type(None)),
"t_axis": (int, type(None)),
"frame_interval_s": (int, float, type(None)),
"t_track_backend": str,
"t_link_threshold": (int, float),
"t_max_displacement_px": (int, float, type(None)),
"t_max_displacement_um": (int, float, type(None)),
"t_project_for_tracking": bool,
"save_original_images": bool,
"keep_intermediate": bool,
"keep_original_images": bool,
"mask_parallel": bool,
"mask_gpu_indices": str,
"watch_folder": bool,
"watch_pipeline": str,
"watch_normalization_pool": str,
"watch_measure_settings": str,
"watch_classify_settings": str,
"watch_settle_seconds": float,
"watch_poll_seconds": float,
"watch_idle_minutes": float,
"microscope_feedback": bool,
"microscope_driver": str,
"microscope_simulated_folder": str,
"microscope_positions": str,
"microscope_stage_transform": list,
"microscope_event_table": str,
"microscope_event_query": str,
"microscope_max_events": int,
"microscope_timepoints": int,
"microscope_interval_seconds": float,
"cloud_anonymous": bool,
"cloud_profile": str,
"cloud_endpoint": str,
"cloud_cache": str,
"cloud_wells": str,
"cloud_fields": int,
"cloud_level": int,
"cloud_results": str,
"save": bool,
"plot": bool,
"tensorboard": bool,
"verbose": bool,
"cell_mask_dim": (int, type(None)),
"cell_min_size": int,
"cytoplasm_min_size": int,
"nucleus_mask_dim": (int, type(None)),
"nucleus_min_size": int,
"pathogen_mask_dim": (int, type(None)),
"pathogen_min_size": int,
"save_png": bool,
"crop_mode": list,
"use_bounding_box": bool,
"png_size": list,
"png_dims": list,
"png_channel_mapping": dict,
"classifier_family": str,
"normalize_by": str,
"save_measurements": bool,
"uninfected": bool,
"dialate_pngs": bool,
"dialate_png_ratios": list,
"cells": list,
"cell_loc": list,
"pathogens": list,
"pathogen_loc": (list, list),
"treatments": list,
"treatment_loc": (list, list),
"channel_of_interest": (int, str, list, type(None)),
"measurement": str,
"nr_imgs": int,
"um_per_pixel": (int, float),
"pathogen_limit": (bool, int),
"nuclei_limit": (bool, int),
"filter_min_max": (list, type(None)),
"channel_dims": list,
"backgrounds": list,
"background": (int, float),
"outline_thickness": int,
"outline_palette": str,
"input_statistics": str,
"input_mean": list,
"input_std": list,
"crop_dtype": str,
"outline_color": str,
"overlay_chans": list,
"normalization_percentiles": list,
"filter": bool,
"fill_in":bool,
"adjust_cells": bool,
"row_limit": int,
"tables": list,
"image_nr": int,
"dot_size": int,
"point_color": str,
"point_alpha": float,
"outline_width": float,
"umap_canvas_width": int,
"umap_sidebar_width": int,
"n_neighbors": int,
"min_dist": float,
"metric": str,
"tsne_perplexity": float,
"tsne_learning_rate": float,
"tsne_early_exaggeration": float,
"tsne_max_iter": int,
"pca_whiten": bool,
"pca_svd_solver": str,
"isomap_n_neighbors": int,
"isomap_path_method": str,
"spectral_affinity": str,
"spectral_n_neighbors": int,
"random_seed": int,
"gpu": bool,
"eps": float,
"min_samples": int,
"batch_correction": str,
"batch_column": str,
"batch_control_column": (str, type(None)),
"batch_control_values": (
str, int, float, list, tuple, type(None),
),
"batch_covariate_column": (str, list, tuple, type(None)),
"batch_combat_mean_only": bool,
"batch_min_samples": int,
"batch_missing_control": str,
"filter_by": (str, type(None)),
"img_zoom": float,
"plot_by_cluster": bool,
"plot_cluster_grids": bool,
"remove_cluster_noise": bool,
"remove_highly_correlated": bool,
"log_data": bool,
"black_background": bool,
"remove_image_canvas": bool,
"plot_outlines": bool,
"plot_points": bool,
"smooth_lines": bool,
"clustering": str,
"exclude": (str, type(None)),
"exclude_grnas": (list, type(None)),
"normalise_fraction": bool,
"image_source": str,
"stream_method": str,
"channel_arrays": list,
"bounding_box": bool,
"object_distances": bool,
"object_distance_maxima": bool,
"object_distance_intensity": bool,
"annotation_source": str,
"fields": (list, str, type(None)),
"barcode_mismatches": int,
"holdout_plate": (str, list, type(None)),
"cell_max_size": (int, float, type(None)),
"nucleus_max_size": (int, float, type(None)),
"pathogen_max_size": (int, float, type(None)),
"cell_area_outlier_mads": (float, int, type(None)),
"nucleus_area_outlier_mads": (float, int, type(None)),
"cell_intensity_outlier_mads": (float, int, type(None)),
"nucleus_intensity_outlier_mads": (float, int, type(None)),
"positive_control_wells": (list, type(None)),
"negative_control_wells": (list, type(None)),
"mixed_control_wells": (list, type(None)),
"col_to_compare": str,
"pos": str,
"neg": str,
"embedding_by_controls": bool,
"plot_images": bool,
"reduction_method": str,
"save_figure": bool,
"color_by": (str, type(None)),
"analyze_clusters": bool,
"resnet_features": bool,
"test_nr": int,
"radial_dist": bool,
"bystander_measurements": bool,
"bystander_reach_in_diameters": float,
"confluency": bool,
"confluency_source": str,
"confluency_channel": (int, type(None)),
"confluency_window": int,
"confluency_qc_threshold": (float, int, type(None)),
"bleach_correction": str,
"measure_gpu": bool,
"measurement_backend": str,
"measurement_backend_target": str,
"profiling": bool,
"profiling_metadata": str,
"profiling_treatment_column": (str, list),
"profiling_negative_control": (str, list),
"profiling_normalization": str,
"profiling_feature_selection": list,
"profiling_correlation_threshold": (float, int),
"profiling_phenotype_column": str,
"profiling_databases": list,
"cell_cycle": bool,
"cell_cycle_method": str,
"cell_cycle_channel": (int, type(None)),
"cell_cycle_gates": (list, type(None)),
"cell_cycle_mitotic_ratio": (float, int, type(None)),
"cell_cycle_fucci_channels": (list, type(None)),
"cell_cycle_labels": str,
"cell_cycle_model": str,
"cell_cycle_epochs": int,
"wound_closure": bool,
"wound_source": str,
"wound_channel": (int, type(None)),
"wound_window": int,
"wound_threshold": (float, int, type(None)),
"wound_hours_per_frame": (float, int, type(None)),
"wound_conditions": dict,
"intensity_calibration": bool,
"intensity_calibration_wells": (list, str, type(None)),
"intensity_calibration_statistic": str,
"intensity_calibration_offset": (float, int),
"plate_barcode_source": str,
"plate_barcodes": (dict, str, type(None)),
"plate_barcode_column": str,
"plate_barcode_token_env": str,
"time_to_event": bool,
"time_to_event_object": str,
"time_to_event_mode": str,
"time_to_event_column": str,
"time_to_event_threshold": (float, int, type(None)),
"time_to_event_persist": int,
"time_to_event_origin": str,
"time_to_event_min_frames": int,
"time_to_event_hours_per_frame": (float, int, type(None)),
"time_to_event_group": str,
"time_to_event_conditions": (list, type(None)),
"time_to_event_reference": str,
"time_to_event_covariates": (list, type(None)),
"viability": bool,
"viability_dead_channel": (int, type(None)),
"viability_live_channel": (int, type(None)),
"viability_thresholds": (list, type(None)),
"viability_negative_wells": (list, str, type(None)),
"viability_positive_wells": (list, str, type(None)),
"viability_plate_map": str,
"cellprofiler_pipeline": str,
"spatial_measurements": bool,
"spatial_neighbor_radius": int,
"calculate_correlation": bool,
"manders_thresholds": list,
"homogeneity": bool,
"homogeneity_distances": list,
"save_arrays": bool,
"cytoplasm": bool,
"merge_edge_pathogen_cells": bool,
"cells_per_well": int,
"pathogen_size_range": list,
"nucleus_size_range": list,
"cell_size_range": list,
"pathogen_intensity_range": list,
"nucleus_intensity_range": list,
"cell_intensity_range": list,
"target_intensity_min": int,
"model_type": str,
"base_model": str,
"mask_src": str,
"test_src": str,
"test_mask_src": str,
"save_path": str,
"channel_axis": (int, type(None)),
"min_train_masks": int,
"max_train_images": (int, type(None)),
"nimg_per_epoch": (int, type(None)),
"nimg_test_per_epoch": (int, type(None)),
"scale_range": float,
"save_every": int,
"save_each": bool,
"heatmap_feature": str,
"grouping": str,
"min_max": str,
"n_estimators": int,
"test_size": float,
"location_column": str,
"positive_control_id": str,
"negative_control_id": str,
"n_repeats": int,
"top_features": int,
"remove_low_variance_features": bool,
"classes": dict,
"class_folder_names": list,
"schedule": str,
"loss_type": str,
"image_size": int,
"crop_size": int,
"epochs": int,
"val_split": float,
"dropout_rate": float,
"init_weights": bool,
"amsgrad": bool,
"use_checkpoint": bool,
"mixed_precision": bool,
"gradient_accumulation_steps": int,
"intermedeate_save": (bool, list, tuple, type(None)),
"pin_memory": bool,
"n_jobs": int,
"ram_guard": bool,
"augment": bool,
"cell_types": list,
"cell_plate_metadata": (list, list),
"pathogen_types": list,
"pathogen_plate_metadata": (list, list),
"treatment_plate_metadata": (list, list),
"cell_chann_dim": (int, type(None)),
"nucleus_chann_dim": (int, type(None)),
"pathogen_chann_dim": (int, type(None)),
"plot_nr": int,
"plot_control": bool,
"remove_background": bool,
"target": str,
"dependent_variable": (str, list),
"regression_panel_manifest": (dict, str, type(None)),
"analysis_mode": str,
"inference": str,
"analysis_unit": str,
"guide_min_wells": (int, list),
"guide_primary_min_wells": (int, type(None)),
"guide_permutations": int,
"guide_permutation_seed": int,
"guide_permutation_block": str,
"guide_nuisance_columns": list,
"grna_statistic": str,
"guide_presence_threshold": (int, float),
"guide_permutation_batch_size": int,
"multiple_testing_method": str,
"fdr_alpha": (int, float),
"p_threshold_alpha": (int, float),
"p_threshold_kind": str,
"rra_alpha": (int, float),
"rra_permutations": int,
"group_lasso_lambda": (int, float, str, type(None)),
"count_grna_column": str,
"count_value_column": str,
"independent_variable_layout": str,
"wide_predictor_columns": list,
"model_data_layout": str,
"tolerance": (int, float),
"invert_dependent_variable": (bool, int),
"score_column": str,
"y_lims": (list, type(None)),
"regression_type": (str, type(None)),
"regression_backend": str,
"alpha": (int, float, str, type(None)),
"model_plate_position": bool,
"random_row_column_effects": bool,
"l1_ratio": float,
"quantile": float,
"hinge_threshold": (float, type(None)),
"hinge_n_boot": int,
"huber_t": float,
"spline_knots": int,
"spline_degree": int,
"lasso_n_boot": int,
"lasso_selection_threshold": float,
"regression_qc": bool,
"transform": (str, type(None)),
"intercept": str,
"intercept_value": float,
"agg_type": str,
"min_cells_per_well": int,
"target_height": (int, type(None)),
"target_width": (int, type(None)),
"rescale": bool,
"resample": bool,
"model_name": str,
"Signal_to_noise": int,
"learning_rate": float,
"weight_decay": float,
"batch_size": int,
"pipeline_style": str,
"batch_fields": int,
"keep_npz": bool,
"n_epochs": int,
"from_scratch": bool,
"width_height": list,
"resize": bool,
"fraction_threshold": float,
"calibrate_fraction_threshold": bool,
"mix":str,
"model_type_ml":str,
"exclude_conditions":list,
"exclude_rows": (dict, type(None)),
"remove_highly_correlated_features":bool,
'file_type':str,
'model_path':str,
'dataset':str,
'score_threshold':float,
'tta_enabled':bool,
'tta_rotations':bool,
'tta_horizontal_flip':bool,
'tta_vertical_flip':bool,
'tta_aggregation':str,
'tta_min_agreement':float,
'tta_max_std':float,
'sample':(int, list, type(None)),
'file_metadata':(str, type(None), list),
"train":bool,
"test":bool,
'train_channels':list,
"optimizer_type":str,
"dataset_mode":str,
"annotated_classes":list,
"annotation_column":str,
"apply_model_to_dataset":bool,
"custom_model": (str, type(None)),
"png_type":str,
"path_string":str,
"crop_source":str,
"object_array":str,
"coordinate_columns":list,
"crop_shape":str,
"custom_model_path":str,
"resume_checkpoint":str,
"generate_training_dataset":bool,
"normalize":bool,
"overlay":bool,
"target_layer":str,
"test_mode":bool,
"dry_run":bool,
'smoothgrad_samples':int,
'smoothgrad_sigma':float,
'occlusion_window':int,
'occlusion_stride':int,
'ig_steps':int,
'ig_baseline':str,
'attribution_steps':int,
'attribution_baseline':str,
'sanity_check':bool,
'counterfactuals':bool,
'counterfactual_crops':int,
'counterfactual_epochs':int,
'counterfactual_condition':str,
'counterfactual_target':str,
'counterfactual_generator':str,
'object_type':str,
"parasite_table": str,
"compartment": str,
"vacuole_key": str,
'replication_method': str,
"vacuole_link_distance": (int, float, type(None)),
"vacuole_link_factor": (int, float),
"parasite_count_column": (str, type(None)),
"max_parasites_per_vacuole": int,
"require_host_cell": bool,
"non_power_of_two_warn": float,
"outside_channel": int,
"total_channel": (int, type(None)),
"intensity_statistic": str,
"background_correction": str,
"outside_threshold_method": str,
"outside_threshold": (float, type(None)),
"stain_baseline_wells": (list, type(None)),
"analysis_excluded_wells": (list, type(None)),
"control_quantile": float,
"min_control_objects": int,
"min_objects_for_threshold": int,
"min_objects_for_bimodality": int,
"bimodality_cutoff": float,
"threshold_agreement_tolerance": float,
"threshold_sensitivity": float,
"inflation_warn": float,
"min_parasites_per_well": int,
"min_parasite_area": (int, float),
"max_parasite_area": (float, type(None)),
"min_total_intensity": (float, type(None)),
"extracellular_class": str,
"seed_wells_from_cells": bool,
"group_column": str,
"level": str,
"change_plate": bool,
"qc_plot_max_panels": int,
"resume":bool,
"resume_search":bool,
"checkpoint_path":(str, type(None)),
"test_images":int,
"remove_background_cell":bool,
"remove_background_nucleus":bool,
"remove_background_pathogen":bool,
"remove_background_organelle":bool,
"organelle_background":(int, float),
"organelle_signal_to_noise":(int, float),
"figuresize":int,
"cmap":str,
"pathogen_model":str,
"cell_model_name":str,
"nucleus_model_name":str,
"pathogen_model_name":str,
"segmentation_backend":str,
"cellpose3_add_nucleus_channel":bool,
"cellpose3_size_model":bool,
"cellpose3_resample":bool,
"cellpose3_augment":bool,
"cellpose3_percentile_low":(int, float),
"cellpose3_percentile_high":(int, float),
"normalize_input":bool,
"filter_column":str,
"target_unique_count":int,
"threshold_multiplier":int,
"threshold_method":str,
"count_data":list,
"score_data":list,
"paired_data":list,
"min_observations_per_hit": int,
"nontargeting_control_grnas":list,
"metadata_files":list,
"filter_value":list,
"x_lim":(list, type(None)),
"log_x":bool,
"log_y":bool,
"reg_alpha":(int,float),
"reg_lambda":(int,float),
"prune_features":bool,
"cross_validation":bool,
"offset_start":int,
"chunk_size":int,
"single_direction":str,
"delete_intermediate":bool,
"outlier_detection":bool,
"CP_prob":int,
"diameter":int,
"flow_threshold":float,
"cell_diameter":int,
"nucleus_diameter":int,
"pathogen_diameter":int,
"diameter_estimate_n_fields":int,
"seg_qc":str,
'image_qc_mode':str,
'image_qc_channels':list,
'image_qc_min_focus':dict,
'object_filters':dict,
'image_qc_max_saturation':dict,
'image_qc_saturation_level':dict,
'image_qc_max_nonfinite':float,
'image_qc_classifier':bool,
'image_qc_classifier_model':(str, type(None)),
'image_qc_classifier_labels':(str, type(None)),
'image_qc_classifier_threshold':(int, float),
"seg_qc_min_objects":int,
"robustness_report": bool,
"robustness_fields": int,
"robustness_crop": (int, type(None)),
"robustness_diameter_factors": (list, str),
"robustness_flow_thresholds": (list, str),
"robustness_cellprob_thresholds": (list, str),
"robustness_enhancement": bool,
"robustness_tolerance": (float, int),
"real_object_classifier": (str, type(None)),
"real_object_threshold": (float, int),
"seg_qc_count_ratio":float,
"seg_qc_size_ratio":float,
"seg_qc_border_fraction":float,
"seg_qc_outlier_mad":float,
"seg_qc_outlier_fraction":float,
"seg_qc_foreground_fraction":float,
"seg_qc_split_ratio":float,
"seg_qc_min_diameter":float,
"seg_qc_tiny_fraction":float,
"seg_qc_max_object_fraction":float,
"seg_qc_plate_fail_fraction":float,
"consolidate":bool,
"distance_gaussian_sigma": (int, type(None)),
"infection_xgb_n_estimators": int,
"infection_xgb_max_depth": int,
"infection_xgb_learning_rate": float,
"infection_xgb_subsample": float,
"infection_xgb_colsample_bytree": float,
"infection_xgb_reg_lambda": float,
"infection_xgb_random_state": int,
"infection_xgb_n_jobs": int,
"infection_xgb_proba_threshold": float,
"infection_xgb_margin": float,
"infection_xgb_top_features": int,
"infection_xgb_proba_column": str,
"infection_xgb_drop_ambiguous": bool,
"infection_xgb_ambiguous_low": float,
"infection_xgb_ambiguous_high": float,
"infection_xgb_min_cells_per_class": int,
"infection_pca_method": str,
"infection_pca_random_state": int,
"motility_ylim": tuple,
"motility_xlim": tuple,
"seconds_per_frame": int,
"pixels_per_um": float,
"infection_intensity_n_bins": int,
"db_table_name": str,
"infection_intensity_qc_graphs": bool,
"infection_intensity_qc_panel_path": str,
"infection_intensity_mode": str,
"infection_intensity_strategy": str,
"infection_intensity_qc": bool,
"straightness_threshold": float,
"drop_straight_tracks": bool,
"track_outlier_zscore": float,
"max_displacement": float,
"tracked_object": str,
"motility_analysis": bool,
"reuse_existing_measurements": bool,
'infection_pca_umap_search': bool,
'infection_pca_umap_n_neighbors_grid':list,
'infection_pca_umap_min_dist_grid':list,
'infection_pca_pathogen_weight':float,
'infection_pca_log_intensity':bool,
'infection_pca_tsne_search':bool,
'infection_pca_tsne_perplexity_grid':list,
'infection_pca_tsne_learning_rate_grid':list,
'infection_intensity_qc_scope': str,
'infection_pca_max_cells':int,
'infection_pca_min_gt_separation':float,
'infection_pca_min_silhouette':float,
'infection_pca_umap_n_neighbors':int,
'infection_pca_umap_min_dist':float,
'infection_pca_tsne_perplexity':float,
'number_of_organelles': int,
'organelle_channel': (int, type(None)),
'organelle_type': str,
'organelle_morphology': str,
'organelle_method': str,
'organelle_diameter': int,
'organelle_model_name':str,
'organelle_remove_border':bool,
'organelle_log_min_sigma': int,
'organelle_log_max_sigma': int,
'organelle_log_num_sigma': int,
'organelle_log_threshold': float,
'organelle_tophat_radius': int,
'organelle_watershed_spots': bool,
'organelle_ridge_sigmas': list,
'organelle_ridge_filter': str,
'organelle_skeletonize': bool,
'organelle_network_threshold':str,
'organelle_adaptive_block_size': int,
'organelle_adaptive_offset': int,
'organelle_morph_radius': int,
'organelle_fill_holes': int,
'organelle_cellprob_threshold': float,
'organelle_flow_threshold': float,
'organelle_resample': bool,
'organelle_mask_dim':(int, type(None)),
'organelle_chann_dim':(int, type(None)),
'organelle_rolling_ball':bool,
'organelle_rolling_ball_radius':int,
'organelle_clahe':bool,
'organelle_clahe_clip_limit':float,
'organelle_mask_within_cells':bool,
'organelle_dog_sigma_low':float,
'organelle_dog_sigma_high':float,
'organelle_hysteresis_low':float,
'organelle_hysteresis_high':float,
'organelle_unet_model_path':str,
'organelle_unet_threshold':float,
'organelle_ring_sigma_inner':float,
'organelle_ring_sigma_outer':float,
'organelle_ring_min_prominence':float,
'organelle_ring_fill_method':str,
'summarize_organelles_by':(str, list, type(None)),
'early_stopping_patience':int,
'class_balance':str,
'strict_errors':(bool, type(None)),
'max_failure_rate':(float, type(None)),
'queue_by_uncertainty':bool,
'queue_measure':str,
'queue_diversity':str,
'queue_limit':int,
'cross_validation_folds':int,
'cross_validation_enabled':bool,
'classifier_evaluation':bool,
'nested_cv_inner_folds':int,
'evaluation_calibration':str,
'evaluation_bins':int,
'evaluation_fail_on_leakage':bool,
'leakage_audit_train_test':bool,
'leakage_hash_content':bool,
'leakage_require_identity':bool,
'generate_full_dataset':bool,
'tar_path':str,
'n_top_examples':int,
'balance_to_smallest':bool,
'write_random_annotation_column':bool,
'cv_group_by':str,
'logit_adjust_tau':float,
'focal_alpha':( float, type(None)),
'focal_gamma':float,
'label_smoothing':float,
'cell_perimeter_fraction':float,
'nucleus_perimeter_fraction':float,
'pathogen_perimeter_fraction':float,
'organelle_perimeter_fraction':float,
'organelle_min_area':int,
'organelle_max_area':(int, type(None)),
'organelle_min_intensity':float,
'organelle_max_intensity':float,
'cell_remove_border_objects':bool,
'nucleus_remove_border_objects':bool,
'pathogen_remove_border_objects':bool,
'organelle_remove_border_objects':bool,
'regex': str,
'target_sequence': str,
'window_length': int,
'column_csv': str,
'barcode_set': (list, tuple, dict, type(None)),
'grna_csv': str,
'row_csv': str,
'save_h5': bool,
'comp_type': str,
'comp_level': int,
'mode': str,
'fill_na': bool,
'class_metadata': list,
'metadata_item_1_name': (str, type(None)),
'metadata_item_1_value': (str, type(None)),
'metadata_item_2_name': (str, type(None)),
'metadata_item_2_value': (str, type(None)),
'size': int,
'test_split': float,
'cam_type': str,
'correlation': bool,
'shuffle': bool,
'class_column': str,
'group_by_class': bool,
'max_area': int,
'max_bins': (int, type(None)),
'min_area_bin': int,
'um_per_px': float,
'grayscale': bool,
'invert': bool,
'percentiles': (list, type(None)),
'plateID': str,
'random_test': bool,
'target_size': int,
'CP_probability': int,
'FT': (int, float),
'circularize': bool,
'nr': int,
'save_dtype': str,
'folders': (list, type(None)),
'csv_name': (str, type(None)),
'data_column': (str, list, type(None)),
'csv': (str, type(None)),
'cv_csv': (str, type(None)),
'data_column_cv': (str, type(None)),
'columnID': (str, type(None)),
'control_sgrnas': (list, type(None)),
'fraction_grna': (str, type(None)),
'scores': (str, type(None)),
'feature_importance': bool,
'permutation_importance': bool,
'shap': bool,
'shap_sample': bool,
'include_all': bool,
'filter_1': (list, type(None)),
'value_col': (str, type(None)),
'threshold': (int, float, str, list, type(None)),
'red_channel': int,
'green_channel': int,
'blue_channel': int,
}
_clone_organelle_registry(expected_types)
#: The background switch of every slot after the first, numbered as the user
#: counts: ``remove_background_organelle_2`` ... (item 76, the maintainer's
#: name, 2026-09-30; lettered ``remove_background_organelleb`` before).
SLOT_BACKGROUND_SWITCHES = tuple(
_background_switch_key(role) for role in ORGANELLE_SLOT_ROLES[1:])
for _key in SLOT_BACKGROUND_SWITCHES:
expected_types.setdefault(_key, bool)
#: The slot prefixes, built ONCE. `str.startswith` takes a tuple and does the
#: whole comparison in C, which is the entire point of hoisting this: the
#: comprehension below used to build `f'{role}_'` inside an `any()` over every
#: role, for every key. MEASURED at import on 2026-09-05: 37,855 keys x 364
#: roles was 13.7 million generator steps and 26.5 million `startswith` calls,
#: 4.1 s of the 6.7 s that importing this module cost -- which is most of why
#: opening a module tripped GNOME's "not responding" dialog. Same answer,
#: same order; see tests/test_settings_imports_fast.py.
_ORGANELLE_SLOT_PREFIXES = tuple(
f'{role}_' for role in ORGANELLE_SLOT_ROLES[1:])
DYNAMIC_ORGANELLE_SETTINGS = frozenset(
key for key in expected_types
if key.startswith(_ORGANELLE_SLOT_PREFIXES))
#: Settings that are declared -- typed here, tooltipped, offered by a GUI
#: category -- but that NOTHING in spaCR reads. Setting one is a silent no-op,
#: which is the worst failure mode there is: the run starts, finishes, and
#: produces a plausible wrong answer, and on a 40-plate cluster job that costs
#: a GPU-week to discover. ``spacr.validate`` turns each of these into a
#: pre-flight ERROR and ``spacr.cli.apply_overrides`` refuses a ``--set`` that
#: names one, both quoting the working spelling below.
#:
#: A key belongs here when its name appears NOWHERE in ``spacr/*.py`` outside
#: the ``expected_types`` / ``tooltips`` / ``descriptions`` / ``categories``
#: literals in this file -- no reader, and no ``setdefault`` either, so no
#: pipeline's own defaults can trip the check.
#: ``tests/test_dead_settings.py`` re-derives the set from the source on every
#: run, so the registry cannot rot in either direction: a key that gains a
#: reader must leave it, and a key that loses its last reader must join it.
#:
#: They stay declared rather than being deleted so that an old settings CSV
#: still loads far enough to be told, by name, what to use instead.
#: The old `crop_source` spellings and what they mean now. ACCEPTED, NOT
#: REFUSED: every settings CSV in existence carries one of them.
_IMAGE_SOURCES = {
"pre_generated": "load_images",
"generate": "load_images",
"png": "load_images",
"load_images": "load_images",
"merged": "stream_images",
"on_demand": "stream_images",
"stream": "stream_images",
"stream_images": "stream_images",
"merged_db": "stream_images",
"auto": "auto",
}
def _canonical_image_source(value) -> str:
"""One of ``load_images`` / ``stream_images`` / ``auto``.
An unrecognised value falls back to loading rather than raising: a
settings file naming a source spaCR never had should still open the
module, and the panel shows what it resolved to.
Two old spellings used to fall into that unrecognised branch and come
back as ``load_images``. ``on_demand`` is the older name for STREAMING,
so a settings file asking for it was answered with the opposite source;
and ``auto`` -- which is what `crops.resolve_crop_source` defaults to,
and means "PNGs if they were exported, merged planes if they were not"
-- was flattened to a fixed choice, so a project with no `data/` folder
stopped finding its own crops. Both are mapped explicitly now, and
``auto`` survives as itself because both readers downstream understand
it.
``merged_db`` (`crops.STREAM_FROM_DB`, the viewers' database stream) is
a STREAMING source, so it resolves to ``stream_images`` -- training has
no database mode, and the unrecognised branch would have answered it
with the opposite direction, as it once did ``on_demand``.
"""
return _IMAGE_SOURCES.get(str(value or "").strip().lower(),
"load_images")
def _coordinate_columns_for(object_array):
"""The coordinate column(s) for an object array, or None.
THROUGH `stream_dataset`, so the derivation and the streamer cannot
disagree about which column an object is identified by.
A LIST, because `coordinate_columns` is declared `list` and has been
since before this derivation existed. Handing back the bare string would
make every run of the module fail its own settings validation -- which
is what it did on the first attempt.
"""
try:
from ._stream_selection import coordinate_column
return [coordinate_column(object_array)]
except Exception: # noqa: BLE001
return None
def _fold_the_classes(settings):
"""Fill `annotation_column` and `class_metadata` from `classes`.
THROUGH `classify_classes`, so the panel, the defaults and the pipeline
cannot disagree about what a class means.
"""
try:
from .classify_classes import fold_into_classes
return fold_into_classes(settings)
except Exception: # noqa: BLE001
return settings
def _outlier_criteria():
"""Return criteria shared by outlier settings and filtering logic.
:returns: the (key, human name) pairs a run can filter outliers on.
READ FROM `_outlier_criteria` AND NOT FROM `outlier_filter`. They are the
same four pairs either way; the difference is that `outlier_filter` needs
pandas, and this function is called while a settings panel is being laid
out. Importing pandas to read four pairs of strings cost 211 ms on the
main thread, which the user sees as the interface stopping.
There is no fallback tuple here any more. There used to be, for the case
where the import failed, and it was a second copy of the same four pairs
that nothing compared with the first -- so a criterion added in one place
and not the other would have been silently dropped from the panel.
"""
from ._outlier_criteria import CRITERIA
return CRITERIA
tooltips = {
'psf_measurement_source': "(str) - original (default) measures normal Measure intensities, including its standard rescaling and registered illumination/preprocessing hooks, without PSF processing. Stored PSF parameters are ignored with this choice. processed applies the calibrated PSF after those stages before quantitative features; select convolve or deconvolve. Source files and exported crops keep their original behavior. The intensity_rescale database table records the choice, kernel identity, calibration and algorithm. Incompatible existing measurements require a separate project/output database.",
'psf_operation': "(str) - Optional point-spread processing of every selected segmentation intensity channel after illumination and before normalization. none is off (default); convolve simulates optical blur; deconvolve uses Richardson–Lucy. Source images and measurement intensities stay original. Changing this setting rebuilds Mask inputs when preprocessing is on; preprocessing off requires an exact completed match. This operates on 2D projected fields, not a 3D optical reconstruction.",
'psf_source': "(str) - gaussian (default) constructs a sampled Gaussian approximation from explicit FWHM and image pixel spacing; it does not estimate microscope optics. measured loads a calibrated TIFF or NPY kernel. A single kernel is applied independently to each selected segmentation channel; use only where its calibration is appropriate for every selected channel.",
'psf_objective': "(str) - Objective used to infer PSF calibration that is left unset: auto reads magnification, numerical aperture, immersion and pixel size from the first source image's OME metadata and otherwise assumes a 20x/0.75 air objective; a table row such as 10x/0.30 air, 40x/0.95 air, 60x/1.40 oil or 100x/1.45 oil uses that objective's values, with a 6.5 µm camera pixel and 520 nm emission unless the metadata says otherwise. Explicit psf_image_sampling_um and psf_fwhm_um always win. Default auto.",
'psf_path': "(str or None) - Default None (unset). Measured 2D PSF TIFF/NPY. Spatial dimensions must be odd; values finite, nonnegative and nonzero. The center pixel is the optical origin. The kernel is normalized to sum to one and captured once per processing run. Its contents and file identity enter psf/segmentation_application.json.",
'psf_image_sampling_um': "(list or None) - Default None (unset). Image pixel spacing [Y, X] in micrometers; both numbers must be positive and finite. Unset with a Gaussian PSF, Mask and timelapse infer it from the first source image's OME/ImageJ calibration, else camera pixel / magnification of psf_objective (6.5 µm / 20 = 0.325 µm by default), and print each value's source. Measure and measured kernels require it explicitly.",
'psf_kernel_sampling_um': "(list or None) - Default None (unset). Measured kernel pixel spacing [Y, X] in micrometers. Must match image sampling; mismatched kernels are refused rather than silently resampled. Unused for a Gaussian approximation.",
'psf_fwhm_um': "(list or None) - Default None (unset). Gaussian full width at half maximum [Y, X] in micrometers; both values must be positive and finite. Unset, Mask and timelapse calculate 0.51 × emission wavelength / NA from image metadata or psf_objective (520 nm, NA 0.75: 0.354 µm by default), an approximation of the ideal widefield PSF, not measured resolution. Measure requires it explicitly.",
'psf_iterations': "(int) - Richardson–Lucy iterations, 1–200; default 20. Higher values may amplify noise and artifacts. Unused for convolution. Processing is cancellable between iterations, uses symmetric boundaries and retains floating point intensities without clipping to the integer source range.",
'unmix': "(bool) - Spectral unmixing: estimate how much of each dye bleeds into the other channels from single-stain control wells, then unmix every field before it is segmented or measured. Make Masks unmixes each raw field across all its channels before illumination correction, the PSF and the enhancement chain; Measure unmixes the measured channels before its preprocessing. The matrix is printed and recorded with the run. Needs unmix_controls. Default False.",
'unmix_controls': "(str) - The single-stain control wells, as channel:well[,well] entries separated by semicolons, for example 0:A01,A02; 1:B01. The channel is the dye's own channel, counted as in the stack or merged array; its wells hold that dye alone. Channels without controls are taken to bleed into nothing. Up to 24 fields per dye are read. Ignored unless unmix is on. Default blank.",
'unmix_background_percentile': "(float) - Percentile of each channel's pixels taken as its background, from 0 up to but not including 100. It is set aside before each field is unmixed and added back after, so a channel with no dye stays at its own background level rather than being pulled below it. Keep it below the fraction of the field that is empty. Default 5.0.",
'n2v_denoise': "(bool) - Self-supervised denoising: train a Noise2Void (N2V2) network per segmentation channel on this run's own noisy fields, with no clean images, and denoise every field with it after illumination correction and before the PSF and the enhancement chain. Needs the CAREamics backend from the Model Zoo; training wants a GPU. The models' hashes and training losses are recorded with the run. Default False.",
'n2v_model': "(str) - Folder of trained Noise2Void models, one channel_<c>.ckpt per segmentation channel, such as the n2v folder of an earlier run on the same microscope. Blank trains new models on up to eight of this run's fields into its own n2v folder and reuses them when the run is resumed. Ignored unless n2v_denoise is on. Default blank.",
'n2v_epochs': "(int) - Training passes over the Noise2Void patches, at least 1. More epochs denoise better up to a point and take longer: on a CPU, 30 epochs over ten 512 x 512 crops took under four minutes, and eight full 2000 x 2000 fields take about twelve times as long per epoch, so train on a GPU. Changing it trains new models. Ignored when n2v_model names trained models. Default 20.",
'enhance_background': "(str) - Background subtraction for every selected segmentation channel after illumination correction and before normalization, the first step of the enhancement chain Make Masks tunes. rolling_ball removes a fitted surface of the radius below and flattens uneven illumination; tophat keeps what is brighter than its surroundings and is faster; none is off. Set the radius larger than the largest object. A resumed run refuses Mask inputs made with a different chain. Default 'none'.",
'enhance_background_radius': "(int) - Radius in pixels of the rolling ball or the top-hat disk. Make it larger than the largest object and smaller than the scale the illumination varies on; a radius under the object size removes the objects with the background. Default 50.",
'enhance_background_scale': "(float) - Fraction of full size the background is estimated at, above 0 and at most 1. The surface is scaled back up before subtraction, so only the estimate is smaller; 1.0 is scikit-image's exact answer and is slow on large fields. Default 0.5.",
'enhance_denoise': "(str) - Smoothing after background subtraction and the PSF, before the contrast curves. gaussian is a Gaussian filter with the strength as sigma in pixels; median removes speckle and keeps edges; bilateral and nlm (non-local means) keep edges better and are slow on whole fields; tv is total-variation (Chambolle), the edge-preserving smoother offered in place of anisotropic diffusion; none is off. Default 'none'.",
'enhance_denoise_strength': "(float) - How much smoothing, above 0: the Gaussian's sigma in pixels, the median's and the bilateral's disk radius, the non-local means cut-off in multiples of the measured noise, or ten times the total-variation weight on 0..1. Default 1.0.",
'enhance_percentile_clip': "(bool) - Clip each selected channel plane to two percentiles of its own intensities before the contrast curves. Removes hot and dead pixels from the range the curves are drawn on without stretching anything, so intensity thresholds keep their units. Default False.",
'enhance_percentile_low': "(float) - Lower percentile of the enhancement chain's percentile clip, at least 0 and below the upper percentile. Pixels darker than this percentile of their own plane are raised to it. Default 1.0.",
'enhance_percentile_high': "(float) - Upper percentile of the enhancement chain's percentile clip, at most 100 and above the lower percentile. Pixels brighter than this percentile of their own plane are lowered to it. Default 99.0.",
'enhance_gamma': "(float) - Exponent the intensities are raised to on 0..1, above 0; 1.0 is off. Below 1 lifts the dim end so faint objects rise out of the background; above 1 pushes it down and keeps only the bright ones. Default 1.0.",
'enhance_log': "(bool) - Logarithmic transform on 0..1, log(1 + gain x) / log(1 + gain), applied after gamma. It compresses the bright end and lifts the dim one, more strongly near zero than a gamma below 1 does. Default False.",
'enhance_log_gain': "(float) - Factor the unit-interval intensities are multiplied by before the logarithm, above 0. Larger compresses the bright end harder; as it goes to zero the curve becomes the identity. Unused unless enhance_log is on. Default 10.0.",
'enhance_sqrt': "(bool) - Square-root transform on 0..1, applied after the logarithm: the curve a gamma of 0.5 draws, offered by name. It lifts the dim end of every selected channel before detection. Default False.",
'enhance_clahe': "(bool) - Contrast-limited adaptive histogram equalisation, per tile rather than over the whole field. Brings out objects in a dim corner without saturating the bright middle, and amplifies noise in empty tiles, which the clip limit bounds. Default False.",
'enhance_clahe_tile': "(int) - Side of one CLAHE tile in pixels, at least 8. Make it larger than one object and smaller than the scale the illumination varies on; a tile the size of one object equalises the object against itself. Default 64.",
'enhance_clahe_clip': "(float) - CLAHE clip limit, above 0 and at most 1. Higher gives more contrast and more amplified noise in tiles that hold only background; unused unless enhance_clahe is on. Default 0.01.",
'enhance_equalize': "(bool) - Global histogram equalisation of each selected channel plane, the strongest of the contrast curves. A field that is mostly background has its background stretched across half the range. Default False.",
'enhance_sharpen': "(bool) - Unsharp mask after the contrast curves. Makes edges steeper so a threshold lands on the boundary; too much amount puts a bright rim around every object and a dark moat outside it. Default False.",
'enhance_sharpen_radius': "(float) - Blur radius in pixels the unsharp mask is built from, above 0: about the scale of the edges to sharpen. Unused unless enhance_sharpen is on. Default 1.0.",
'enhance_sharpen_amount': "(float) - How much of the unsharp mask is added back, at least 0. One is a normal sharpen; above 2 the halos around objects start to become objects of their own. Default 1.0.",
'image_source':
"(str) - Source of classification images. 'load_images' reads "
"previously exported object crops; 'stream_images' generates crops "
"from merged image and mask arrays during training. Streaming avoids "
"creating a separate export for each object, channel, and crop-shape "
"combination. Default 'load_images'.",
'stream_method':
"(str) - Method used to locate objects for streamed crops. 'column' "
"uses coordinates stored in the object table and requires "
"object_array and channel_arrays. 'array' uses labelled objects in "
"the same object_array plane and additionally requires "
"bounding_box. "
"Settings that do not apply to the selected method are ignored. "
"Default 'column'.",
'channel_arrays':
"(list[int]) - Zero-based intensity-plane indices included in each "
"streamed image, in output-channel order. This setting applies to "
"both stream methods; changing the order changes the channel mapping "
"presented to the model. Default [0, 1, 2].",
'bounding_box':
"(bool) - Crop geometry used when stream_method is 'array'. True "
"retains the rectangular region enclosing each labelled object, "
"including local background and neighbouring signal. False retains "
"only pixels within the object mask and sets surrounding pixels to "
"zero. Default True.",
'holdout_plate':
"(str | list | None) - Train without this plate and score on it. "
"None -- the default -- splits within the available data. "
"Cross-validation can otherwise learn plate-associated variation "
"rather than the phenotype while retaining apparently strong "
"metrics. Performance on a held-out plate directly evaluates "
"generalization across plates. The run is rejected if either split "
"would omit a class.",
'barcode_mismatches':
"(int) - How many mismatched bases a barcode may carry and still be "
"matched. 0, the default and prior behavior, "
"requires an exact match, so a single sequencing error anywhere in a "
"barcode discards the read. A read that falls within the mismatch "
"budget of two barcodes remains unassigned because its source "
"barcode cannot be resolved without risking guide misattribution.",
'cell_max_size':
"(int | None) - Drop cells larger than this many pixels. None, the "
"default, disables the filter and preserves prior behavior. The "
"minimum sizes remove debris; only a maximum removes a segmentation "
"artifact, which passes every minimum and carries its area into the "
"classifier and the regression. The run prints how many objects each "
"bound dropped.",
'nucleus_max_size':
"(int | None) - Drop nuclei larger than this many pixels, counted "
"the same way nucleus_min_size counts them. None -- the default -- "
"disables it and preserves prior behavior. A minimum removes "
"debris; only a maximum removes nuclei merged into a single mask, "
"which pass every minimum and then bias downstream area and "
"DNA-content measurements. The run "
"prints how many objects each bound dropped.",
'pathogen_max_size':
"(int | None) - Drop pathogens larger than this many pixels, counted "
"the same way pathogen_min_size counts them. None -- the default -- "
"disables it and preserves prior behavior. This bound removes a "
"vacuole of tightly packed parasites segmented as one "
"object: it passes every minimum size and inflates both the per-cell "
"burden and the mean parasite area. The run prints how many objects "
"each bound dropped.",
'fields':
"(list | str | None) - Fields to process, enabling selected fields "
"to be reprocessed without repeating the complete plate. None, the default, "
"-- processes every field found. A list or a comma-separated string "
"of field ids in any spelling spaCR accepts ('f3', 3, 'F003'), or a "
"glob such as 'f1*'. The run writes only the fields named, into the "
"same folders, so a re-run replaces those and leaves the rest.",
'object_distances':
"(bool) - Measure every distance between and within objects: centre "
"to centre, centre to the nearest surface of each other object type "
"(both directions), surface to surface -- which is zero when two "
"objects touch and is what 'how far apart are they' means -- the "
"overlap fraction, how far the centre sits from its own boundary, "
"and how close the object is to the edge of the field. On by "
"default: the calculation is computationally expensive on a 3-D field, but measuring "
"again afterwards costs more. The cost is one "
"distance transform per object type per field, not one per pair of "
"objects.",
'object_distance_maxima':
"(bool) - Also find the intensity maxima inside each object and "
"measure where they are: how many, how spread out, and how far each "
"is from the object's own boundary, its centre, and the nearest "
"surface of every other object type. The most expensive part of "
"object_distances, and ignored when that is off. Default True.",
'object_distance_intensity':
"(bool) - Include the families that need the intensity images: the "
"local maxima above and the scalar displacement between each "
"channel's intensity-weighted centre of mass and the geometric "
"centroid. False measures geometry only. Ignored when "
"object_distances is off. Default True.",
'annotation_source':
"(str) - Organism annotation joined to regression results. "
"'toxoplasma' uses bundled Toxoplasma gondii tables without a network. "
"Other organism names or NCBI taxon IDs fetch UniProt entries; a "
"single accession such as P04637 fetches that entry. Results are "
"cached beside the outputs for offline reuse. An unrecognised name "
"leaves results unannotated and suggests similar names. Leave empty "
"for no annotation. Default 'toxoplasma'.",
'cell_area_outlier_mads':
"(float | None) - Optionally remove objects whose cell area exceeds "
"this many scaled "
"median absolute deviations from the median. Filtering occurs before "
"guide fractions are computed, so removed objects do not contribute "
"to normalization. MAD-based limits are robust to skewed area "
"distributions. Set to None to disable; a value of 5 is a conservative "
"starting point. Default None.",
'nucleus_area_outlier_mads':
"(float | None) - Scaled median-absolute-deviation threshold for "
"nucleus area before guide annotation. Lower values remove more "
"nucleus-size extremes before guide fractions and normalization are "
"computed. Set to None to disable. Default None.",
'cell_intensity_outlier_mads':
"(float | None) - Optionally apply the scaled-MAD filter to "
"cell-channel intensity "
"before guide annotation. This can exclude extreme measurements from "
"overexposure or bright debris. Set to None to disable. Default None.",
'nucleus_intensity_outlier_mads':
"(float | None) - Scaled median-absolute-deviation threshold for "
"nucleus-channel intensity before guide annotation. Lower values "
"remove more extreme measurements, including saturated nuclei or "
"bright debris. Set to None to disable. Default None.",
'response_speed':
"(str) - Inference-effort level requested from the provider for each reply. "
"'fast' minimizes latency and is sufficient for straightforward "
"setting questions; 'deep' uses more of the provider quota and is "
"intended for traceback or run-summary analysis. "
"The three levels map onto whatever each vendor CLI calls them, so "
"the same choice means the same thing across providers. "
"Default 'balanced'.",
'auto_file_issues':
"(bool) - Report a failed run as an issue on the public spaCR GitHub "
"repository. The report carries the traceback, the run's settings "
"and software versions, with paths, login and host names and "
"credentials redacted. With issue reporting set to 'always' (the "
"default) it is filed automatically, once per error, when GitHub is "
"signed in; with 'ask' it opens in a preview and is sent only when "
"you press Send report. Default True.",
'route_errors_through_ai':
"(bool) - Send a failing run's traceback to the AI Console automatically "
"and request an explanation. The traceback and "
"the settings leave your machine when this is on; leave it off if "
"your paths or filenames are themselves sensitive. Default True.",
'console_aware':
"(bool) - Include new console output and complete tracebacks as "
"context when a question is sent through the AI Console. This adds "
"execution details without manual copying. The attached console "
"content is sent to the configured provider; disable this setting "
"when output may contain sensitive information. Default True.",
'system_prompt':
"(str) - Text prepended to every conversation as persistent instructions, "
"such as the organism, plate conventions and requested answer format. "
"Empty uses spaCR's own prompt unchanged. "
"Default ''.",
"image_type": "(str) - Exported crop folder to read: 'cell_png', 'nucleus_png', 'pathogen_png', or 'cytoplasm_png'. This setting applies only to pre-generated image loading and is not used when crops are streamed from merged/. Default 'cell_png'.",
"crop_size": "(int) - How many pixels across each cell is drawn. One number: the crop is square. Larger fills the tab with fewer cells per page; the pagination follows it. Default 200.",
"normalize_channels": "(bool/list) - Percentile-stretch each channel before drawing, so a dim stain is visible beside a bright one. Uses 'percentiles'. This changes only the displayed image; the stored crop and all measurements remain unchanged. Default None (off).",
"outline": "(bool/list) - Draw the object's own outline over the crop, as in the annotation app. The outline is computed before any channel is zeroed so it cannot trace a channel that is no longer displayed. Default None (off).",
"outline_threshold_factor": "(float) - How aggressively the outline is cut from the object channel. Above 1 tightens the outline onto the brightest core; below 1 loosens it outward. Only read when 'outline' is on. Default 1.25.",
"outline_sigma": "(float) - Gaussian blur applied before the outline is found, in pixels. Larger gives a smoother, less speckled boundary at the cost of fine detail. Only read when 'outline' is on. Default 4.",
"edge_thickness": "(float) - Outline width as a fraction of object size rather than a fixed pixel count, preserving relative width across crop_size values. Larger values improve boundary visibility in small thumbnails but cover more interior pixels. Default 0.1.",
"edge_transparency": "(float) - Outline opacity on a 0-100 scale: 0 hides the outline and 100 makes it fully opaque. Intermediate values blend the outline with the image. Default 100.",
"edge_image": "(bool) - Draw the object outline over the source image. False draws the outline on a blank background so that boundary geometry can be evaluated independently of image intensity. Default False.",
"object_size": "(int/list) - The smallest object, in pixels, that is outlined at all. Debris below it is skipped rather than traced. Default (0, 0) — no minimum.",
"show_all_in_well": "(bool) - Display every cell in the well and highlight the selected cells, rather than displaying selected cells alone. Highlighting matches the annotation app. When disabled, only cells already attributed to the guide are shown, so every visible cell is selected and the visible fraction is necessarily 1. Disable this setting only when isolated selected cells are required. Default True.",
"half_widths": "(float) - Set the score-window half-width in robust scales (1.4826 × median absolute deviation) on either side of the score implied by the coefficient. This value applies to every coefficient. It changes the range described in the montage caption, not the number of cells shown; the displayed count comes from guide-fraction estimates so narrow windows do not leave wells underrepresented. Default 1.0.",
"baseline": "(str) - Reference used to compute a coefficient's implied score: 'screen_median' (all wells), 'control_median' (non-targeting controls), or 'zero'. Use the control baseline when a global screen shift makes the screen median unsuitable as the reference. Default 'screen_median'.",
"cap": "(int) - Maximum number of objects drawn for one coefficient across all wells. This limit prevents impractically large montages; the caption reports when the limit is applied and how many objects are omitted. Memory use and page count increase in proportion to this value: at 2000, a well tab contains about 280 MB of thumbnails across 33-83 pages. Default 2000.",
"cell_picking": "(str) - How displayed cells are selected. 'rank' takes the highest scores, with the count determined by the inferred fraction. 'attributed' assigns each cell a guide probability and takes those above the threshold. 'assigned' assigns exactly one guide to every cell in the well, so each guide receives the number of cells implied by its reads. 'multivariate' uses all measurements rather than the score alone and requires the gene-by-measurement sweep effects grid. Default 'rank'.",
"picking_threshold": "(float) - The probability a cell must reach before it is called for this guide, for the 'attributed' and 'multivariate' pickers. Below it a cell is ambiguous and is not shown. The other pickers compute no probability and ignore it. Default 0.55.",
"dst": "(str) - Folder receiving versioned tables, manifests and figures. Default '' uses a module-specific folder beside the primary input, keeping different analyses separated.",
"db_path": "(str) - Exact measurements.db whose objects an analysis reads. Choosing another database is refused when object or crop identities do not match, preventing cross-experiment joins. Default '' requires selection.",
"predictions_file": "(str) - Existing per-object prediction CSV joined to measured objects. Analysis modules read this exact output and do not silently rerun a model or substitute another run. Default '' requires selection.",
"path_column": "(str) - Column in a prediction CSV containing crop paths for a one-to-one object join. Change it only when the exporter used another name. Default path.",
"batch_correction": "(str) - Plate/batch correction applied before Image UMAP, ML screen classification or phenotype regression. 'none' leaves measurements alone; 'center' removes each plate's mean shift; 'zscore' aligns plate means and variances; 'robust_zscore' uses median/MAD and tolerates outliers; 'combat' models the batch effect while protecting the terms named in batch_covariate_column. Correct when plates were stained or imaged separately; leave off when they were not, since every method removes real signal that happens to align with plate. See spacr.batch_correction.correct_batch_effects. Default 'none'.",
"batch_column": "(str) - Metadata column that identifies independent acquisition batches, normally 'plateID'. Every analyzed row must have a value and at least batch_min_samples rows must occur in each batch. Use an acquisition date or instrument ID only if that is the nuisance source you intend to remove. Default 'plateID'. API: spacr.batch_correction.correct_batch_effects.",
"batch_control_column": "(str or None) - Metadata column containing reference-control labels for control_center, normally 'columnID' for plate controls. It is ignored by center, zscore, robust_zscore, and none. Blank follows col_to_compare in Image UMAP or location_column in Classify (ML); regression defaults to 'columnID'. API: spacr.batch_correction.correct_batch_effects.",
"batch_control_values": "(str, number, list or None) - Reference/negative-control value(s) in batch_control_column used by control_center. Each plate needs at least batch_min_samples matching rows. Image UMAP falls back to neg and Classify (ML) to negative_control_id when this field is blank; regression requires an explicit value. Default varies by module. API: spacr.batch_correction.correct_batch_effects.",
"batch_covariate_column": "(str, list or None) - Metadata column(s) naming the biological effects ComBat must preserve, for example 'condition' or 'condition,timepoint'. ComBat estimates the batch effect from residuals after fitting these terms, so unlisted effects may be removed with the plate effect. Include every treatment effect that must remain in the corrected data. See spacr.batch_correction.correct_batch_effects. Default None.",
"batch_combat_mean_only": "(bool) - True corrects only the additive batch shift and leaves each batch's scale alone. Use it when the plates differ in level but not in spread, or when a batch has too few rows for a stable variance estimate. False (the default) corrects both location and scale, which is standard ComBat. Ignored by every method other than combat. API: spacr.batch_correction.correct_batch_effects.",
"batch_min_samples": "(int) - Minimum number of rows required in every batch, and minimum matching reference controls per batch for control_center. Correction stops with an actionable error below this threshold because a one- or two-object plate estimate is unstable. Default 3. API: spacr.batch_correction.correct_batch_effects.",
"batch_missing_control": "(str) - Policy when control_center cannot find enough reference controls on a plate: 'error' stops rather than silently mixing corrected and raw plates; 'skip' leaves that plate unchanged and records a warning. Default 'error'. API: spacr.batch_correction.correct_batch_effects.",
"threshold_direction": "(list, list-of-lists, int or None) - Which side of 'threshold' to keep when prefiltering objects for annotation: 'higher' keeps rows whose measurement is >= the threshold, 'lower' keeps rows <= it. Give one value, or one per entry in 'measurement' (a single string is broadcast to the whole list). Default 'higher'.",
"threshold": "(list, list-of-lists, int or None) - Cut-off applied to 'measurement' before the annotation grid loads, so you only label the objects you care about. Accepts a number or a quantile code 'q1'-'q9' (q3 = the 30th percentile of that column), or one entry per measurement when measurement is a list. Empty or None loads every object unfiltered. Default 2000 where a numeric cutoff is used; empty where the setting is optional.",
"cell_model_name": "(str) - Cell-segmentation weights. Cellpose 4 provides the stock 'cpsam' model; alternatively, provide a CPSAM checkpoint created by Train Cellpose, loaded as pretrained_model. Legacy names ('cyto', 'cyto2', 'cyto3', 'nuclei') resolve to cpsam because Cellpose 4 no longer ships those models -- unless segmentation_backend is 'cellpose3', which runs the real model of that name. Only diameter changes inference (scaling by 30/diameter); model_type and diam_mean are not used in v4.0.1+. Default 'cpsam'.",
"nucleus_model_name": "(str) - Weights used to segment nuclei. Valid values are 'cpsam' or a path to a custom CPSAM checkpoint produced by Train Cellpose. 'nuclei' and 'nucleus' map to 'cpsam' because Cellpose 4 removed the pre-SAM models -- unless segmentation_backend is 'cellpose3', which runs the real Cellpose 3 'nuclei'. Configure nucleus_diameter to control scale; of the three parameters that once distinguished models only diameter still acts (eval rescales by 30/diameter); model_type and diam_mean are dropped. Default 'cpsam'.",
"pathogen_model_name": "(str) - Which weights segment pathogens. 'cpsam' or a path to your own Train Cellpose checkpoint. The bundled toxo_pv_lumen / toxo_cyto checkpoints are Cellpose-3 CPnet, which CPSAM cannot load, so they map to 'cpsam' and are reported; with segmentation_backend 'cellpose3' a Cellpose 3 name or checkpoint path runs as written. The older 'pathogen_model' key still overrides this one when set. Of the three parameters that used to distinguish models only diameter still acts (eval rescales by 30/diameter); model_type and diam_mean are logged 'not used in v4.0.1+' and dropped. Default 'cpsam'.",
"segmentation_backend": "(str) - Which model segments cells, nuclei and pathogens; masks from different models are not comparable. 'cellpose' (default) runs each object's model name. 'cellpose3' runs Cellpose 3 with an object's cyto3, cyto2, cyto or nuclei model name, or a Cellpose 3 checkpoint's path; other names mean nuclei for nuclei and cyto3 otherwise. 'samcell' and 'dinocell' are 2-D models for live-cell and label-free images that read only the object's channel. Every backend but 'cellpose' installs from the Model Zoo into its own environment and refuses z_stack and t_stack runs. Default 'cellpose'.",
"cellpose3_add_nucleus_channel": "(bool) - Cellpose 3 reads two channels as [cyto, nucleus]. When True, a cell segmented by a Cellpose 3 model also receives the nucleus channel as its second channel, as cyto, cyto2 and cyto3 were trained; when False, the model sees the cell channel alone. Read only for objects whose model is a Cellpose 3 model. Default True.",
"cellpose3_size_model": "(bool) - When True, a named Cellpose 3 model estimates each image's object diameter with its own size model, which is slower and, on Toxoplasma vacuoles, far less accurate; when False, the object's diameter setting is used, or its magnification default when blank. Read only for Cellpose 3 models. Default False.",
"cellpose3_resample": "(bool) - When True, Cellpose 3 computes flows at the image's original size before building masks, which gives smoother outlines at some cost in time; when False, it builds them at the rescaled size. Read only for objects whose model is a Cellpose 3 model. Default True.",
"cellpose3_augment": "(bool) - When True, Cellpose 3 averages its prediction over flipped copies of overlapping tiles, which is slower and can steady borderline objects. It stands in for net averaging, which Cellpose 3.1 no longer offers. Read only for objects whose model is a Cellpose 3 model. Default False.",
"cellpose3_percentile_low": "(float) - Lower percentile of Cellpose 3's per-channel normalization: this intensity becomes 0 before the model sees the image. Raising it darkens more of the background. Cellpose 3's models were trained with 1. Read only for Cellpose 3 models. Default 1.0.",
"cellpose3_percentile_high": "(float) - Upper percentile of Cellpose 3's per-channel normalization: this intensity becomes 1 before the model sees the image. Lowering it brightens dim objects and saturates bright ones. Cellpose 3's models were trained with 99. Read only for Cellpose 3 models. Default 99.0.",
"cell_diameter": "(int or None) - Expected cell diameter in pixels. Cellpose 4 rescales the image by 30/diameter before segmentation, aligning the expected object size with the scale used to train CPSAM; leave it None to segment at native scale. Set it when cells are much larger or smaller than ~30 px and segmentation produces fragmented or merged masks. spacr.diameter.estimate_diameters estimates a value from the selected fields. Default None.",
"nucleus_diameter": "(int or None) - Expected nucleus diameter in pixels, used by Cellpose 4 to rescale the image by 30/diameter before segmentation. None segments at native scale. Because nuclei are commonly the smallest segmented objects, this parameter often requires explicit configuration for low-magnification acquisitions. spacr.diameter.estimate_diameters estimates a value. Default None.",
"pathogen_diameter": "(int or None) - Expected pathogen diameter in pixels, used by Cellpose 4 to rescale the image by 30/diameter before segmenting. None segments at native scale. Intracellular parasites are often only a few pixels across at low magnification, where rescaling matters most. spacr.diameter.estimate_diameters proposes a value. Default None.",
"diameter_estimate_n_fields": "(int) - How many fields spacr.diameter.estimate_diameters reads before it proposes cell_diameter, nucleus_diameter and pathogen_diameter from blob statistics instead of requiring manual estimation. Fields are taken on an even stride across the sorted plate, so rows and columns are both represented rather than the first few wells; each field costs about a second of CPU and loads neither torch nor Cellpose. Increase it to 10–20 when wells are heterogeneous or confidence is low; decrease it to 2–3 for a faster preliminary estimate. Default 5.",
'image_qc_mode': '(str) - Image screening before segmentation. off preserves the normal run; report saves metrics and flags without excluding anything; exclude skips flagged fields under the saved policy. No images are deleted and excluded fields are not reported as zero-object results. Reports: qc/image_quality.json and .csv. Default off.',
'image_qc_channels': '(list) - Acquisition-channel identifiers to screen before Mask. Empty means every stored raw channel. These are zero-based array channels in v1 and mapped acquisition-channel identifiers in v2. Threshold dictionaries use the same identifiers. Default [].',
'object_filters': "(dict) - Object filters on any scalar scikit-image regionprop, per object type, for example {'cell': [{'property': 'area', 'min': 200}, {'property': 'solidity', 'min': 0.9}]}. An object is kept when min <= value <= max; a missing side is off. Intensity properties read the object's own raw channel. An area minimum is also Cellpose's min_size. An old file's cell_min_area and the other retired per-object bounds become rows here when it loads. Default {}.",
'image_qc_min_focus': '(dict) - Minimum acceptable raw Laplacian variance by channel, for example {0: 25.0, 2: 10.0}. Empty disables focus exclusions. Calibrate using representative fields from the same acquisition; units are intensity squared. For a volume, the best-focus plane is used so defocused neighboring z planes alone do not reject it. Default {}.',
'image_qc_max_saturation': '(dict) - Largest allowed saturated-pixel fraction by channel, for example {2: 0.01}. Values range from 0 to 1. Saturation uses the acquisition level or integer dtype ceiling, never the brightest observed pixel. Empty disables saturation exclusions. Default {}.',
'image_qc_saturation_level': '(dict) - Acquisition saturation level by channel, for example {0: 4095, 2: 65535}. Set 4095 for a 12-bit detector stored in uint16. Missing integer levels use the dtype ceiling; floating images require an explicit level if saturation exclusion is enabled. Default {}.',
'image_qc_max_nonfinite': '(float) - Maximum fraction of NaN or infinite pixels allowed per screened channel. Exceeding it flags the field; exclusion occurs only in exclude mode. Range 0-1; default 0.',
'image_qc_classifier': '(bool) - Also screen each field with a small neural network that flags out-of-focus, saturated, debris-covered, bubble and empty fields, alongside the focus and saturation rules. Flags go into the image-quality report with a probability per class; in exclude mode a flagged field is kept out of segmentation like any other. Runs on the CPU. The built-in model is trained once on synthetic fields with planted defects and cached. Ignored when image_qc_mode is off. Default False.',
'image_qc_classifier_model': '(str or None) - A classifier saved by an earlier run (qc/image_qc_model.pt) to use instead of the built-in one. It is read as tensors only. Blank uses the built-in model. Default None.',
'image_qc_classifier_labels': '(str or None) - Table of hand-labelled fields to fine-tune the classifier on, with a field column (the raw file name) and a label column: good, or out_of_focus, saturated, debris, bubble or empty, several separated by semicolons; an optional channel column picks the channel. With 10 or more fields, cross-validated precision and recall against the rule metrics go to qc/image_qc_benchmark.csv; the tuned model is saved as qc/image_qc_model.pt. Default None.',
'image_qc_classifier_threshold': '(float) - Smallest class probability, from 0 to 1, at which the classifier flags a field. Raise it to flag fewer fields, lower it to miss fewer defects. Default 0.5.',
"seg_qc": "(str) - Segmentation quality control performed when masks are written, before measurement. 'off' skips scoring; 'report' scores every field, writes qc/segmentation_qc_<object>.csv, and displays detected quality issues; 'flag' also writes per-field JSON for downstream processing; 'stop' raises when the plate verdict is 'fail', after writing the scorecard. No mode deletes or omits a field, and 'stop' does not raise for a 'warn' verdict. Default 'report'.",
"robustness_report": "(bool) - After the masks are made, re-segment a few sampled fields with the diameter, the flow and cell-probability thresholds and contrast enhancement each moved a little, and report how much the object count, median area, mean object intensity and the objects themselves change. Settings whose results move more than robustness_tolerance are flagged fragile. Writes qc/segmentation_robustness_<object>.csv and a heatmap. Two-dimensional Cellpose-SAM runs only. Default False.",
"robustness_fields": "(int) - How many fields the robustness report samples at random (seeded by random_seed) and re-segments at every grid point. More fields give a steadier verdict; each field costs one segmentation per grid point. Default 4.",
"robustness_crop": "(int or None) - Side in pixels of the centre crop the robustness report cuts from each sampled field, to keep the grid fast, on a CPU especially. Blank or 0 re-segments whole fields. Default 512.",
"robustness_diameter_factors": "(list) - Multiples of the object diameter the robustness report tries, one at a time; a blank diameter counts as Cellpose's nominal 30 pixels. Default [0.75, 1.25].",
"robustness_flow_thresholds": "(list) - Flow thresholds the robustness report tries in place of the run's own, one at a time. Higher keeps objects whose flows are less consistent. Default [0.2, 0.6].",
"robustness_cellprob_thresholds": "(list) - Cell-probability thresholds the robustness report tries in place of the run's own, one at a time. Lower grows objects and finds faint ones; higher shrinks or drops them. Default [-2.0, 2.0].",
"robustness_enhancement": "(bool) - Also re-segment each sampled field after contrast-limited adaptive histogram equalisation (CLAHE), to see whether contrast enhancement changes what the model finds. Default True.",
"real_object_classifier": "(str or None) - A trained real / not-real classifier, or a folder holding cell.joblib, nucleus.joblib and pathogen.joblib, as written by tools/train_real_object_classifier.py. After the masks are made, each object is cut from the field with the cell, pathogen and nucleus channels and erased from its mask when the classifier calls it not real; the other ids are kept. Verdicts go to qc/real_object_filter_<object>.csv. Not applied to timelapse runs. Default None.",
"real_object_threshold": "(float) - Probability of being real below which the real / not-real classifier erases an object. Higher removes more objects. Only used with real_object_classifier. Default 0.5.",
"robustness_tolerance": "(float) - Largest median relative change in object count, median area or mean intensity, or share of the run's objects not found again, that still counts as stable. A grid point beyond it is flagged fragile. Default 0.2.",
"seg_qc_min_objects": "(int) - Fields with fewer objects than this are classified as near-empty, and robust per-field size statistics are suppressed because the median absolute deviation is unstable for very small samples. Increase the value for confluent cell plates expected to contain hundreds of objects per field; reduce it to 3-5 for low-multiplicity pathogen assays in which few objects per field are expected. Default 10.",
"seg_qc_count_ratio": "(float) - Permitted ratio between a field's object count and the plate median before the field is flagged. A value of 0.25 flags counts below one quarter of the median or above its reciprocal, four times the median. Calibrate this threshold with representative control plates when expected object density varies by assay. Default 0.25.",
"seg_qc_size_ratio": "(float) - Fold change in a field's median object diameter, measured against the plate median, that marks it as fused or fragmented when its object count has moved in the opposite direction. Merging two equal objects into one increases equivalent diameter by a factor of approximately 1.41, while dividing one object into two produces the reciprocal change; the default therefore reflects the expected geometric ratio. Default 1.4.",
"seg_qc_border_fraction": "(float) - Fraction of a field's objects allowed to touch the image edge before the field is flagged. Edge objects are truncated, so their crop dimensions and measured areas are biased downward. Geometry alone places approximately two object diameters on the border, corresponding to about 8% for 60 px cells in a 1400 px field; the default is therefore above the fraction expected in a valid field. Default 0.3.",
"seg_qc_outlier_mad": "(float) - Number of robust standard deviations, each defined as 1.4826 times the median absolute deviation, that an object's diameter may differ from the field median before it is classified as a size outlier. Median and MAD limit the influence of debris. The default of five accommodates heavier-tailed biological size distributions; a threshold of three can flag valid objects. Default 5.",
"seg_qc_outlier_fraction": "(float) - Fraction of a field's objects that must fall outside the robust size range before the field is reported as containing multiple size populations. Such objects commonly represent debris, fused pairs, or fragments. Decrease the value to increase sensitivity to mixed fields, at the cost of more flags. Default 0.15.",
"seg_qc_foreground_fraction": "(float) - Foreground coverage at or above which a field is classified as confluent. The distance-transform fusion check runs only above this threshold. Increasing it reduces computation but decreases sensitivity to fusion in moderately dense fields; decreasing it evaluates more sparse fields and increases runtime. It matches the fused_fraction used by the diameter estimator. Default 0.35.",
"seg_qc_split_ratio": "(float) - Minimum ratio of distance-transform maxima to mask objects required to flag fusion in a field already classified as confluent. A value of 2 requires at least two resolved maxima per mask object on average. Increasing the value reduces sensitivity to fused masks. Default 2.",
"seg_qc_min_diameter": "(float) - Equivalent diameter in pixels below which an object is treated as a fragment; it drives the over-segmentation check and sets the seed floor of the fusion cross-check. Lower it to two or three for punctate organelles, where five-pixel objects may represent valid signal rather than debris, and raise it for large cells where components of that size are likely segmentation fragments. Default 5.",
"seg_qc_tiny_fraction": "(float) - Fraction of a field's objects that may be smaller than seg_qc_min_diameter before the field is classified as over-segmented. Dividing one cell into multiple fragments increases this fraction, whereas a valid field containing limited debris should remain below the default threshold. Default 0.3.",
"seg_qc_max_object_fraction": "(float) - Fraction of the field that a single label may cover before it is classified as evidence of fusion rather than a valid object. A component covering one quarter of a field commonly represents a confluent monolayer merged into one mask; the diameter estimator excludes such components for the same reason. Lower it for small objects on large fields; raise it only when a single large object per field is expected. Default 0.25.",
"seg_qc_plate_fail_fraction": "(float) - Fraction of failing fields at which the plate-level scorecard changes from warn to fail. The default 0.1 corresponds approximately to one column of a 96-well plate. This setting changes only the reported verdict and does not determine which fields are processed. Default 0.1.",
"nucleus_cellprob_threshold": "(float) - Cellpose cell-probability threshold for the nucleus channel, passed straight to model.eval as cellprob_threshold. A pixel must exceed it to join a mask, so raising it shrinks masks and drops dim nuclei, while lowering it grows masks and recovers faint ones along with more debris. Useful range about -6 to 6; default 0.",
"pathogen_cellprob_threshold": "(float) - Cellpose cellprob_threshold for the pathogen channel: a pixel is claimed by a mask only if its predicted object probability exceeds this. Lower it (toward -6) to recover dim or small parasites and grow mask boundaries; raise it (toward 6) to shrink masks and drop faint objects. Useful range about -6 to 6. Default -1.",
"nucleus_flow_threshold": "(float) - Cellpose flow_threshold for nucleus masks: the maximum allowed error between a mask's recomputed flows and the network's predicted flows. Lowering it discards more irregularly shaped nuclei, giving fewer but cleaner objects; raising it keeps nearly everything Cellpose proposes. Typical range 0 to 3; above 3 practically every candidate is kept. Default 0.4, Cellpose's own default.",
"pathogen_flow_threshold": "(float) - Cellpose flow_threshold for pathogen masks: a candidate mask is discarded when its recomputed flows disagree with the network prediction by more than this. Raise it to keep more, sometimes misshapen, parasites; lower it to keep only clean, well-formed objects. Typical range 0.0-3.0; above 3 practically every candidate is kept. Default 0.4, Cellpose's own default.",
"cell_channel": "(int or None) - Zero-indexed raw acquisition channel that Cellpose segments into cell masks; it also selects which channel the cell_background, cell_signal_to_noise and remove_background_cell settings are applied to during preprocessing. Set to None and no cell masks, cell table or cell crops are produced. At least one of cell/nucleus/pathogen/organelle_channel must be an integer or the run aborts. Default None.",
"nucleus_channel": "(int or None) - Zero-indexed raw acquisition channel segmented into nucleus masks, and the channel that nucleus_background, nucleus_signal_to_noise and remove_background_nucleus apply to. None means no nucleus masks, hence no nucleus table, no cell-to-nucleus linking, and nothing subtracted from the cytoplasm mask. Set it whenever a DNA stain was acquired. Default None.",
"pathogen_channel": "(int or None) - Zero-indexed raw acquisition channel segmented into pathogen masks (Toxoplasma etc.), and the channel pathogen_background, pathogen_signal_to_noise and remove_background_pathogen apply to. None disables pathogen segmentation, the pathogen table, the infected-only filter (uninfected) and the adjust_cells step, which needs cell, nucleus and pathogen masks together. Default None.",
"nucleus_mask_dim": "(int) - Position along the last axis of each merged/*.npy array where the nucleus label mask sits, one plane after the cell mask. With the default four image channels (0-3) that is 5; keep a different number of channels and it shifts by the same amount. None makes measure_crop skip nucleus measurements and cell-to-nucleus linking. Default 5.",
"batch_size": "(int) - How many images are held and processed together in one pass: field stacks during normalization and Cellpose segmentation, crops per step during classifier training and activation maps. Raising it speeds runs up but increases RAM/VRAM roughly linearly; lower it on out-of-memory errors. Defaults: 50 for mask generation, 64 for training.",
"pipeline_style": "(str) - Which mask pipeline runs. 'v1' is the disk-based chain (rename, per-channel folders, npy, npz, mask npy, merged/) that measure, annotate and every downstream tool expect, and is the fully tested path. 'v2' streams from the originals and writes one npy per field with masks appended in place, using roughly 60-80% less disk but producing no .npz. Default 'v1'.",
"batch_fields": "(int) - Streaming pipeline only (pipeline_style='v2'): how many whole field stacks are loaded into RAM before one Cellpose batch is segmented. Larger values keep the GPU busier and cut the number of read passes over the plate, at a memory cost of roughly one full field stack each. Ignored entirely by the v1 pipeline. Default 8.",
"keep_npz": "(bool) - Streaming pipeline only (pipeline_style='v2'): write each in-memory NPZ batch under merged/_scratch/ instead of discarding it, so intermediate data from a failed run can be inspected. This increases disk usage; enable it only for diagnosis. Default False.",
"CP_probability": "(int) - Cellpose cellprob_threshold used by the standalone apply/test-model submodules, where it carries this name instead of the per-object <object>_cellprob_threshold used by the Mask module. Only pixels whose predicted cell probability exceeds it join a mask, so raising it shrinks outlines and drops faint objects while lowering it grows them and recovers dim ones. Default 0.",
"FT": "(float) - Cellpose flow_threshold for the standalone apply/test-model submodules, the counterpart of the Mask module's per-object <object>_flow_threshold. Masks whose recomputed flows disagree with the network's prediction by more than this are discarded, so a low value strips ragged or implausible objects and also loses real ones. Typical range 0 to 3; above 3 practically every candidate is kept. Default 0.4, Cellpose's own default.",
"circularize": "(bool) - Replace each detected mask with an equal-area circle centred on its centroid before measurement. This can reduce boundary variation for approximately circular objects with noisy segmentation outlines. Do not enable it when shape is an outcome, because circularization removes elongation and other morphological differences. Default False.",
"class_column": "(str) - Column containing the per-object class label used by class-proportion analysis. Missing values are filled with 0 rather than dropping the corresponding rows; selecting an incorrect column can therefore assign class zero to every object without raising an error. The value is also appended to the condition when group_by_class is enabled. Default 'test'. Replication starts with the predictions column instead.",
"class_metadata": "(list of lists) - One inner list per training class, holding the metadata values that select that class's objects, for example [['c1'],['c2']] for a two-class run keyed on column. Order fixes the class indices the model learns, so reordering the inner lists relabels the whole training set. Values that occur in no row make the generator select nothing and stop. Default [['c1'], ['c2']].",
"fill_na": "(bool) - When a barcode does not match any entry in its reference CSV, count it under its raw sequence instead of dropping it. When disabled, grouping discards unmatched reads without warning; a reference in the incorrect orientation can therefore produce a reduced table rather than an empty one. Enable this setting to quantify the unmapped fraction. Default False.",
"group_by_class": "(bool) - Whether the endodyogeny condition labels are split by class before proportions are computed: on, the condition string has the class_column value appended, so each condition-class combination becomes its own group; off, classes are pooled within a condition. It changes what the bars count, not how the statistics are weighted. Default False.",
"max_area": "(int) - Upper area cutoff applied before the endodyogeny bins are built; larger objects are excluded. The cutoff is applied after um_per_px scaling and therefore uses square micrometres when a scale is set or square pixels when it is None. The default is effectively unbounded. Default 1000000000.",
"max_bins": "(int or None) - Maximum number of area bins in the endodyogeny histogram. None derives the bin count from the data range and min_area_bin; an integer truncates the range so that the largest objects are pooled into the final bin. Set an integer when plates with different size ranges must share one axis. Default None.",
"metadata_item_1_name": "(str or None) - Name given to the first extra metadata grouping written alongside each generated training crop, for example 'nc' and 'pc' for negative and positive controls. It only labels the group; the values that select the rows are metadata_item_1_value. None writes no extra grouping at all. Default None.",
"metadata_item_1_value": "(str or None) - Metadata values selecting the rows assigned to metadata_item_1_name, for example [['c19','c2'],['c3','c4']]. Values absent from the table contribute no crops. Because the generator reports only the final count, an incorrect value can be indistinguishable from a legitimately rare class. None disables the grouping. Default None.",
"metadata_item_2_name": "(str or None) - Name of the second extra metadata grouping for generated training crops, used when one grouping is not enough to describe the design, for example a treatment axis crossed with the control axis of item 1. Purely a label; metadata_item_2_value selects the rows. None writes no second grouping. Default None.",
"metadata_item_2_value": "(str or None) - The metadata values selecting rows for metadata_item_2_name, in the same nested form as metadata_item_1_value. It is matched independently of item 1, so a row can belong to both groupings at once and will then be written under each. None disables the second grouping. Default None.",
"min_area_bin": "(int) - Width of the smallest area bin in the endodyogeny histogram, and therefore the resolution at which small parasites are distinguished from one another. Expressed in the same unit as max_area, so it follows um_per_px when a scale is set. Too small a value produces sparse noisy bins; too large merges real division states. Default 500.",
"nr": "(int) - Number of example fields shown in a preprocessing diagnostic figure. It does not affect arrays written to disk; increasing it only adds plotting time. Use one field on large plates when a preliminary validation view is sufficient. Default 1.",
"plateID": "(str) - Plate name stamped onto count and score rows that carry no plate of their own, and used as the first field of the plate_row_column key that joins the two tables. It is ignored with a warning when the input already contains more than one distinct plate, so it matters only for single-plate inputs. Default 'plate1'.",
"save_dtype": "(str) - NumPy data type used to write preprocessed image arrays. 'uint16' preserves the full range of a typical microscope camera; conversion to 'uint8' reduces storage by approximately 75% but permanently discards intensity resolution used by downstream measurements. Conversion of integer input to a floating-point type increases storage without preserving additional source information. Default 'uint16'.",
"size": "(int) - Edge length in pixels of each square crop written into the generated training dataset. It must match the classifier input at training time, and increasing it later cannot recover detail discarded during cropping. Select a size that preserves the smallest relevant object. Default 224.",
"target_size": "(int) - Edge length in pixels to which training images and masks are resized before Cellpose fine-tuning, applied to both axes to produce square input. Larger values preserve finer boundary detail while increasing VRAM use and computation approximately quadratically; smaller values reduce computation but may blur segmentation boundaries. Default 1000.",
"test_split": "(float) - Fraction of the generated crops held out as the test set, between 0 and 1. The split respects the grouping level chosen elsewhere, so crops from one well do not straddle it and the score is not inflated by the model recognising the well. Raising it buys a steadier estimate and costs training data. Default 0.1.",
"um_per_px": "(float or None) - Physical size of one pixel, used to convert the endodyogeny area column into square microns before binning. Set it and max_area, min_area_bin and every reported area are in microns; leave it None and they stay in pixels, which makes numbers from objectives of different magnification incomparable. Default 0.1.",
"cell_flow_threshold": "(float) - Cellpose flow_threshold: the maximum allowed error between a candidate mask's recomputed flows and the network's predicted flows. Masks above it are discarded, so lowering it strips ragged or implausible cells but also loses real ones; raising it keeps more. Usable range about 0-3; above 3 practically every candidate is kept. Default 0.4, Cellpose's own default.",
"cell_cellprob_threshold": "(float) - Cellpose cellprob_threshold: only pixels whose predicted cell probability exceeds it are assigned to a mask. Raise it to shrink outlines and drop faint or spurious cells; lower it to grow outlines and recover dim ones. Valid range roughly -6 to 6, default 0. Lower it first when whole cells are missing.",
"channels": "(list of int) - Zero-indexed image channels kept in merged/*.npy and measured by measure_crop; each entry produces its own <object>_channel_<n>_* intensity columns. The list length fixes where masks land, so cell/nucleus/pathogen_mask_dim must shift if you change it. Preprocessing silently resets it to range(n) when it does not match the number of channel folders found. Default [0,1,2,3]. External Masks starts with []; there an empty list means every detected intensity channel, not no channels.",
"crop_mode": "(list) - Mask used to center each PNG crop: 'cell', 'nucleus', 'pathogen', 'cytoplasm' or 'organelle'. One crop set is written per entry into <mode>_png/ folders, so ['cell','nucleus'] doubles the images written and the rows added to png_list. A single png_size such as [224,224] is broadcast to every mode, as are dialate_pngs and dialate_png_ratios; use lists only when modes require different values. A list shorter than crop_mode reuses its final entry for the remaining modes and records a warning. Default ['cell'].",
"custom_regex": "(str or None) - Python regex with named groups that extracts metadata from raw image filenames. It must supply wellID, fieldID and chanID; plateID is optional (falling back to the source folder name), and timeID or sliceID may be absent. With metadata_type='custom', a filename that does not match or lacks a required group is skipped with a warning, which can reduce the dataset. With 'auto', the regex is tried first for Yokogawa renaming and requires only wellID; automatic detection is used if it fails. Default None.",
"cytoplasm": "(bool) - Derive a cytoplasm object per cell (cell mask with nucleus, pathogen and organelle pixels removed) and write it to its own cytoplasm table, which recruitment ratios such as pathogen/cytoplasm intensity are computed from. Requires a cell mask; measure_crop switches it on automatically whenever cell_mask_dim is set, so the value you enter is usually overridden. Default True.",
"diameter": "(float) - Deprecated expected object diameter in pixels, passed to model.eval(diameter=...) by the mask-finetune tool and check_cellpose_models. Cellpose rescales each image by 30/diameter to match its approximately 30-pixel working size; a value below the true diameter upscales the image, whereas a larger value downscales it. Prefer the per-object diameter settings. Default 30.",
"filter": "(bool) - Legacy switch for the old post-Cellpose cleanup pass, which re-ran size/intensity/border filtering and logged '_after_filtration' object counts to the database. The current Cellpose-SAM segmentation path never reads it, so toggling it changes nothing; use the per-object <object>_min_area, <object>_max_area and <object>_perimeter_fraction settings instead. Default False.",
"magnification": "(int) - Objective magnification, used only to derive expected object sizes: pixel diameter is 2*mag+80 for cells, 0.75*mag+45 for nuclei and mag for pathogens, with min/max area limits of diameter^2/4 and diameter^2*10. Explicit cell_diameter, nucleus_diameter or pathogen_diameter override it. Set this to the acquisition objective magnification (10, 20, 40 or 60). Default 40.",
"plaque_model": "(str) - Cellpose checkpoint used to segment plaques: a model-zoo key downloads a checksum-verified checkpoint on first use ('toxoplasma_plaque_v2' is cpsam_plaque_r5, trained on the curated v5 set), a filesystem path selects a custom checkpoint, and 'bundled' is the historical packaged model, which is a Cellpose 3 checkpoint and needs Cellpose 3. Changing it can change every plaque count, so recorded runs should keep the chosen value. Default 'toxoplasma_plaque_v2'.",
"plaque_mode": "(str) - What Plaque Assay reads. 'plaque' takes images that each show one plaque field (a well or a cropped plaque image), segments the plaques and writes per_image and per_plaque tables. 'figure' takes published figures: the YOLO detector finds the plaque images in each figure, the text around them is read (panel letter, column and row labels) and keyed to the figure legend, each image is annotated with its condition and its plaques are segmented, into plaque_figures/plaque_figures.db. Default 'plaque'.",
"figure_detector": "(str) - Figure mode: model-zoo key or checkpoint path of the YOLO detector that finds plaque images inside a figure. Default 'toxoplasma_well_detector_v2', which was measured to find cropped plaque panels in published figures.",
"figure_imgsz": "(str) - Figure mode: comma-separated detector input sizes, all of which are asked and their boxes merged. 640 finds whole faint panels, 1280 finds small dilution spots; neither alone finds both. Default '640,1280'.",
"figure_confidence": "(float) - Figure mode: minimum detector score, 0 to 1, for a box to count as a plaque image. Default 0.25.",
"figure_read_text": "(bool) - Figure mode: read the text printed around each plaque image (needs RapidOCR, part of spacr[papers]) to annotate it with its panel and condition. False names each image by its figure, row and column only. Default True.",
"confirm_annotations": "(bool) - Figure mode: measure only images whose condition a person approved in the Figure preview (saved to figure_annotations.csv in the source folder); the others are counted as waiting. Default False.",
"text_reach_above": '(float) - Figure mode, text detection: How far above the grid of plaque images a column header may sit, in image heights. Raise it when headers are printed well above the images; lower it when the header of a neighbouring panel is picked up. Default 1.0.',
"text_reach_left": '(float) - Figure mode, text detection: How far left of the grid a row label may sit, in image widths. Raise it when row labels are set far to the left; lower it when text from the panel on the left is picked up. Default 1.0.',
"text_reach_below": '(float) - Figure mode, text detection: How far below the grid text may sit and still count as a label, in image heights. Default 0.5.',
"text_use_above": '(bool) - Figure mode, text detection: Use the column header printed above each image as part of its condition. Default True.',
"text_use_left": '(bool) - Figure mode, text detection: Use the row label printed to the left of each image as part of its condition. Default True.',
"text_use_below": '(bool) - Figure mode, text detection: Use text printed under the images as part of the condition. Turn off when captions or axis labels sit under the images. Default True.',
"text_panel_reach": "(float) - Figure mode, text detection: How far up and left of a grid's top-left corner the panel letter (A, B, C...) may sit, in image sizes. Raise it when the letter is set far from the images; lower it when another panel's letter is taken. Default 1.0.",
"text_min_confidence": '(float) - Figure mode, text detection: Ignore OCR words the reader scored below this (0-1). Raise it when stray specks are read as text. Default 0.0.',
"text_ignore": "(str) - Figure mode, text detection: Comma-separated regular expressions; a word matching any is not used as a label, for example scale bars and axis numbers: '^\\\\d+$, [uμ]m$'. Default ''.",
"text_order": "(str) - Figure mode, text detection: Which labels come first when they are joined into one condition: a comma-separated order of above, left and below. Default 'above,left,below'.",
"text_separator": "(str) - Figure mode, text detection: What joins the labels into one condition. Default ' / '.",
"text_reread": '(bool) - Figure mode, text detection: Read the text around each grid of images a second time, enlarged, which finds small or rotated labels the first reading missed. Default True.',
"text_reread_scale": '(int) - Figure mode, text detection: How many times to enlarge the text around each grid for the second reading. Default 3.',
"well_detection": "(str or bool) - Split a plate image into detected wells before plaque segmentation. False passes each source image through whole; True selects the default YOLO detector, while a model-zoo key or checkpoint path selects another detector. Enabling it changes result rows from one per image to one per detected well. Default False.",
"well_confidence": "(float) - Minimum YOLO confidence, from 0 to 1, for keeping a detected well when well_detection is enabled. Raising it removes uncertain boxes but can lose an entire condition; lowering it retains more candidates and can create spurious well crops. Default 0.25.",
"well_pad": "(int) - Extra image pixels retained on every side of a detected well crop, clipped at the source-image boundary. Increase it when the detector box trims the well edge; excessive padding can include neighbouring wells or background. Default 0.",
"plate_format": "(str or None) - Standard culture-plate format used as the physical ruler for detected wells: '6-well', '12-well', '24-well', '48-well' or '96-well'. It converts plaque areas from pixels to square millimetres; None leaves physical-area columns empty unless well_diameter_mm is supplied. Default None.",
"plaque_estimate_growth": "(bool) - Optional experimental plaque time/scale suggestions, saved separately from measurements. Uses the largest quarter of plaques and an assumed linear growth relation. Defaults off. The RH/HFF seven-day reference has about 40 hours held-out endpoint error and has not been validated across times or other conditions. Without a known scale or time, the reference duration is explicitly assumed.",
"plaque_growth_reference_um": "(float) - Reference median equivalent diameter in micrometers of the largest 25 percent of plaques. Default 893.8178699548309, derived from three independent RH/HFF vehicle-control experiments in Kelsen et al. 2023 S20 Data, DOI 10.1371/journal.pbio.3002110. Replace with a matched local reference when available.",
"plaque_growth_reference_hours": "(float) - Positive formation time in hours corresponding to the growth reference diameter. Default 168 (seven days). Linear diameter growth through zero is assumed; this is not a fitted temporal growth curve.",
"plaque_pixels_per_um": "(float, int or None) - Known pixels per micrometer in the analyzed image. Positive values override detected rulers. Leave blank for automatic scale-bar or well-diameter calibration. Per-well values entered in Figure preview take precedence. Default None.",
"plaque_formation_hours": "(float, int or None) - Elapsed plaque formation time in hours, recorded as experimental metadata. Zero is permitted; blank means unknown. Figure preview allows per-well overrides. Default None.",
"colony_counting": "(bool) - Count bacterial or fungal colonies on plate or dish photos instead of segmenting plaques. Each image is one plate, or one well per detected well when well_detection is on; the dish is found by its outline otherwise. Colonies are thresholded against the agar, touching ones are split, and the count, CFU/mL, colony areas and diameters go to colonies/colonies.db in src. No plaque model is loaded. Plaque mode only: Figure mode ignores it. Default False.",
"colony_dilution": "(float, int, dict or str) - Dilution factor of the plated suspension: 10000 for a 10^-4 dilution; fractions such as 0.0001 are inverted. CFU/mL = colonies x dilution factor / plated volume. Enter one number, a dict keyed by filename or stem, or a UTF-8 CSV path with file,dilution columns. Exact filenames take priority over stems; unmatched plates get no CFU/mL. CSV factors must be positive and finite, with unique file identifiers. Default 1.",
"colony_plated_volume_ul": "(float) - Volume of the dilution spread on each plate, in microlitres, the denominator of CFU/mL. Change it with the plating protocol: 100 for a standard spread plate, 1000 for a pour plate of 1 mL. Default 100.",
"colony_too_many": "(int or None) - Plates with more colonies than this are flagged 'too many to count' (TNTC) in per_plate: neighbouring colonies merge and compete, so the count underestimates what was plated. Their CFU/mL is still written, so filter on the flag. Blank turns the check off. Default 300.",
"colony_too_few": "(int or None) - Plates with fewer colonies than this are flagged 'too few to count' (TFTC) in per_plate: so few colonies carry a sampling error too large for the CFU/mL they imply. Their CFU/mL is still written, so filter on the flag. Blank turns the check off. Default 30.",
"colony_polarity": "(str) - Whether colonies are brighter than the agar (bright: white or cream colonies on blood, chocolate or dark agar, or any plate photographed on a dark background) or darker (dark: on a light box or on pale agar). auto tries both and keeps the one whose round objects stand further above the agar. Set it when auto picks the wrong one on a sparse plate. Default auto.",
"colony_threshold": "(float) - How far above the agar a pixel must be to count as colony, in multiples of the agar's own noise. Lower finds faint, small or translucent colonies but also picks up agar texture, bubbles and glare; higher keeps only clear colonies. Default 4.0.",
"colony_min_area_px": "(float, int or None) - Smallest colony counted, in pixels of the original photo. Raise it to ignore dust, bubbles and pinpoint artefacts; lower it for pinpoint colonies. Blank uses 0.4 % of the dish diameter, squared: about 36 pixels for a dish 1500 pixels across. Default None.",
"colony_detector": "(str or None) - A YOLO colony detector to find the colonies with, in place of thresholding: a checkpoint path or a Model Zoo key. It needs the ultralytics package. Detection counts colonies in chains and at the dish rim that thresholding merges or loses; polarity, threshold and minimum area are then not used. Blank thresholds. Default None.",
"well_diameter_mm": "(float, int or None) - Known interior diameter of a detected well in millimetres, overriding plate_format when both are set. It converts the detected pixel diameter into pixels per millimetre and therefore rescales every physical plaque area; use None when the diameter is unknown. Default None.",
"metadata_type": "(str) - Raw-image filename convention, grouped by microscope vendor. Default 'cellvoyager' (Yokogawa CV7000/CV8000). 'custom' uses custom_regex; 'auto' first renames files to Yokogawa naming, using custom_regex when supplied or automatic detection. Provisional conventions come from public-dataset filenames, not vendor documentation. A wrong choice can misassign plate, well, field or channel IDs and channel folders. Use Test on my folder before running.",
"n_jobs": "(int) - CPU workers for parallel stages: measurement, mask adjustment, DataLoader loading, and the sklearn/UMAP calls where -1 means every core. Raise it to shorten CPU-bound steps until RAM or disk I/O saturates. Note the measure-and-crop pipeline overrides your value with cpu_count()-4. Defaults vary by pipeline: cpu_count()-4, -1, or None.",
"ram_guard": "(bool) - Keep worker pools from filling RAM. Before any module starts its workers, spaCR estimates each worker's memory from one input (a field, mask, crop batch or table) and lowers n_jobs to the count that leaves 12.5% of RAM free, printing a warning; the GUI asks first. Off keeps your n_jobs, and Measure fields still wait while free RAM is below that reserve. Default True.",
"normalize_by": "(str) - Percentile source used to rescale cropped PNGs, and only active when 'normalize' is a [low, high] percentile pair: 'png' stretches each crop to its own percentiles, maximising per-object contrast; 'fov' uses percentiles from the whole field, keeping brightness comparable between objects. Choose 'fov' if crop intensities will be compared. Default 'png'.",
"nuclei_limit": '(int, bool, or None) - Cap on nuclei per cell, applied when the per-object tables are merged. None disables the filter, True keeps only single-nucleus cells, and an integer N keeps cells with N or fewer. Cells over the cap are dropped from the merged table entirely. Do not pass False: it is interpreted as 0 and removes every cell, leaving an empty analysis rather than raising an error. Default None. Merged Classifier starts at True and Recruitment starts at 1, so both initially retain only single-nucleus cells. Replication starts at 10.',
"pathogen_limit": "(int, bool, or None) - Maximum pathogens per cell. True or 1 = single pathogen only; None or False = no limit; int = custom limit. Default varies by module (1, 3, 10 or 1000 depending on the factory that fills it), so check the module's own settings rather than assuming one value.",
"masks": "(bool) - Run Cellpose segmentation for every configured object channel (cell, nucleus, pathogen, organelle) and write label stacks to masks/<object>_mask_stack. False performs preprocessing only, producing normalized arrays without label masks; downstream measurement therefore requires a subsequent segmentation step. Default True.",
"delete_intermediate": "(bool) - Legacy force-cleanup switch. True overrides keep_intermediate and keep_original_images, removing stack/, masks/, the numeric per-channel folders, and the orig/ raw backup after merged/ is built. Cleanup is already the default; enable this setting only when cleanup must override those retention settings. Deletion is skipped unless every field of view reached merged/. Default False.",
"save": '(bool or list of bool) - Controls whether the current module writes its optional disk artifacts, such as masks, figures or result tables. Mask accepts a three-item list for [cell, nucleus, pathogen] independently; other modules use one boolean. Default varies by module.',
"reduction_method": "(str) - Dimensionality reduction run before clustering and plotting: 'umap' preserves more global structure and can be fitted on controls then applied to all data, 'tsne' emphasises local neighbourhoods and cannot reuse a fitted model. With 'tsne', min_dist is ignored and n_neighbors is used as perplexity. Anything else raises ValueError. Default 'umap'.",
"test_size": "(float) - Fraction of labelled single-object rows reserved as the test split for the tabular machine-learning classifier; the remainder trains the model. Increasing the value reduces uncertainty in the accuracy estimate but leaves fewer labelled rows for training. Valid range 0-1; default 0.2 (20% test).",
"merge_pathogens": "(bool) - Legacy option that merged two touching pathogen labels into one when their shared boundary exceeded 66% of the smaller object's perimeter, so a single PV split by Cellpose counted once. The current Cellpose-SAM path ignores it - use pathogen_perimeter_fraction instead. Default True.",
"resize": "(bool or float) - Resize every image to target_height x target_width before running Cellpose, then scale the returned mask back to the original dimensions with nearest-neighbour interpolation so measurements remain in original pixels. Enable this setting to match oversized fields to the model's training scale or reduce GPU memory use. Requires target_height and target_width. Default False (True for plaque analysis).",
"embedding_by_controls": "(bool) - Fit the reducer only on control wells - rows whose col_to_compare value equals pos or neg - and then project every object into that space. Use it when the axes should be defined by the control phenotypes so treatments are read relative to them; False fits on all objects. Default False.",
"cam_type": "(str) - Which attribution map is computed. 'gradcam' weights target_layer feature maps by their pooled gradients; 'saliency_image' and 'saliency_channel' map the input gradient, summed or per stain. Also selectable: 'torchcam_gradcam'/'torchcam_gradcam_pp', 'hirescam', 'ablation_cam', 'gradient_shap'/'deeplift_shap', 'saliency' (SmoothGrad) and 'chefer' (ViT relevance). Methods that do not fit model_type are greyed with the reason. Default 'gradcam'.",
"target_layer": "(str) - Dotted attribute path to the convolutional layer whose activations and gradients Grad-CAM hooks, e.g. 'base_model.blocks.3.layers.1.layers.MBconv.layers.conv_b'; utils.recommend_target_layers(model) lists valid names. Later layers give class-specific but coarse maps, earlier ones finer detail. Required for 'gradcam'/'gradcam_pp' - it is auto-filled only when model_type is exactly 'maxvit', and left None it raises. Default None.",
"shuffle": "(bool) - Shuffle the tar dataset in the DataLoader when generating activation maps, so each batch-grid PDF contains a mixed sample rather than consecutive files from one plate or class. False preserves deterministic file order and permits direct alignment with the dataset listing. Default True.",
"correlation": "(bool) - Correlate every input channel with every activation-map channel per image and write the result to the <cam_type>_correlations table: a Pearson coefficient plus Manders M1/M2 at each manders_thresholds percentile (15, 50, and 75 by default). This provides quantitative evidence of stain-specific model attention beyond visual heatmap inspection. save=True is required to write the results to the database. Default True.",
"mode": "(str) - Read-pairing strategy for barcode extraction: 'paired' locates target_sequence in R1 and in the reverse complement of R2 and merges them base-by-base into a quality-weighted consensus; 'single' scans one mate alone, chosen by single_direction. Paired calls barcodes more accurately but discards any read whose anchor is missing from either mate. Default 'paired'.",
"window_length": "(int) - Number of bases sliced out of each read starting at offset_start relative to the target_sequence hit; this window is what the regex is matched against. It must span the whole barcode block (column + gRNA + row) or the regex stops matching and reads are dropped; shorter reads are padded with 'N'. Default 89.",
"infection_intensity_qc_scope": "(str) - Whether infection QC is fitted once or per group: 'combined'/'global'/'all' fits one model on everything, 'plate'/'per_plate' one per plateID, 'well'/'per_well' one per plate-well, and 'none'/'off' skips QC; an unrecognised string falls back to combined behaviour with a warning. Per-well fitting absorbs staining and exposure differences but needs enough cells per well; every group still writes its own QC plot, only the QC payload embedded in the summary panel is taken from the first processed group. Default 'per_well'.",
"adjust_cells": "(bool) - After segmentation, merge cell labels that divide a single pathogen or nucleus, and absorb an anucleate cell fragment into the neighbouring label with which it shares the largest perimeter. Requires cell, nucleus, and pathogen channels and is skipped for timelapse runs. Enable when large infected cells are systematically fragmented by segmentation. Default True.",
"agg_type": "(str) - How per-object scores are collapsed to one value per well before regression: 'mean', 'median', 'quantile' (75th percentile), or None to skip aggregation and regress on individual objects. Median resists a handful of extreme cells; None keeps power but ignores within-well correlation. Forced to a per-well sum for poisson and to None for quantile. Default 'mean'.",
"alpha": "(float) - Regularisation strength for penalised models only: the L1 penalty for 'lasso', the L2 penalty for 'ridge', the combined penalty for 'elasticnet', and the inverse margin for 'hinge'. Larger values shrink more coefficients toward zero. Set it to 'auto' or None to select the value by five-fold cross-validation; the default 1 may over-regularise fraction-scale designs. Other model families reject a non-default alpha rather than ignoring it. Default 1.",
"z_stack": '(bool) - When True, spaCR requires the array to contain an explicit z dimension and raises an error instead of inferring the axis; this enables z_segmentation_mode, anisotropy and stitch_threshold. Standard ingestion collapses z by maximum-intensity projection while organising raw files, so its output has no z axis to segment; supply volumetric arrays directly to spacr.zstack instead. When False, no z-stack code runs and masks match a two-dimensional run. Default False.',
"z_segmentation_mode": "(str) - How the z dimension is handled. The three modes answer different questions and their masks are not comparable, so the choice is recorded alongside them. 'project' collapses the stack with z_projection and segments one plane; it is the only mode the Measure module can consume. 'stitch' segments each plane in 2-D and links labels through the stack. 'volumetric' segments the 3-D volume directly and requires anisotropy or voxel sizes. Default 'project'.",
"z_axis": "(int or None) - Axis of the incoming array that holds z, specified as 0, 1 or 2. None infers it from shape only when one axis is clearly shorter than the other two, such as a 21x512x512 or 512x512x21 stack. An ambiguous shape such as 64x64x64 raises an error because an incorrect axis segments a transposed volume and produces invalid masks. Set this explicitly whenever the acquisition shape is ambiguous. Default None.",
"z_projection": "(str or None) - Method used to collapse z when z_segmentation_mode is 'project'. 'max' retains the brightest value along the stack and is appropriate for sparse fluorescent objects; 'mean' suppresses noise but dilutes signal present in few planes; 'sum' preserves total signal; and 'best_focus' retains only the sharpest plane, which is preferable when one plane is in focus and a maximum-intensity projection would include substantial out-of-focus signal. Ignored by the other modes. Default 'max'.",
"anisotropy": "(float or None) - Ratio of z step to xy pixel size (dz / dxy), used by volumetric mode to represent inter-plane distance. A value of 1.0 on a confocal stack whose z step is 3-10 times the xy pixel size can fuse objects along z. Leave this None and set voxel_size_z_um / voxel_size_xy_um to derive it; if neither is available, volumetric mode raises an error rather than assuming 1.0. Measure also uses it for 3-D region properties and distance transforms. Default None.",
"voxel_size_z_um": "(float or None) - Spacing between consecutive z planes in micrometres, obtained from the acquisition metadata. Together with voxel_size_xy_um it determines anisotropy and converts object volumes from voxel counts into cubic micrometres. Changing it rescales every physical z quantity and the anisotropy used for segmentation; it has no effect on a 'project' run. Measure uses the pair to report 3-D morphology in physical units and records the units in measurement_units. Default None.",
"voxel_size_xy_um": "(float or None) - Width of one pixel in micrometres in the image plane, assumed square. Used with voxel_size_z_um to derive anisotropy and to turn voxel counts into physical volumes and surface areas. Note this is a different setting from um_per_pixel, which only sizes the scale bar drawn on figures and never reaches a measurement. This one does reach measurements, but only on a 3-D run: a 2-D run never applies it, because doing so would turn every *_area from px2 into um2 under an unchanged column name. Default None.",
"stitch_threshold": "(float) - Minimum overlap, as an intersection-over-union between 0 and 1, for a label in one plane to be treated as the same object as a label in the plane below when z_segmentation_mode is 'stitch'. Raising it splits objects that drift or change shape between planes into several shorter ones; lowering it fuses neighbouring objects that merely overlap in projection. Matching is one-to-one, so when two objects both overlap the same object below only the better match inherits its label and the other starts a new one. Ignored by the other two modes. Default 0.25.",
"t_stack": '(bool) - When enabled, spaCR requires each field to be a (T, Z, Y, X) volume. Standard image ingestion collapses z by maximum-intensity projection, so a run using that path stops with an error instead of pretending the projected data are 4-D. Enable this setting only when passing volumes to spacr.zstack.segment_4d through the Python API, and specify t_axis_order because shape alone cannot identify the time axis. When disabled, no 4-D processing occurs. Default False.',
"t_axis_order": "(str or None) - Which of the two leading axes is time and which is z: 'TZYX' for a stack per timepoint, 'ZTYX' for a time series per plane. Real microscopes write both and the array shape cannot distinguish them, so spaCR raises an error until the axis order is specified. An incorrect value does not raise an exception; it links objects at corresponding lateral positions in adjacent z planes and interprets those displacements as temporal motion, producing invalid velocity estimates. Verify the setting against the acquisition axis order. Default None.",
"t_axis": "(int or None) - Index of the time axis in the incoming array, as an alternative to spelling out the whole order in t_axis_order; the z axis is then taken to be the other of the two leading axes, or whatever z_axis says. Use it for an acquisition whose axes are not in either of the two standard orders. When both this and t_axis_order are set they must agree, and spaCR stops if they do not rather than silently preferring one. Default None.",
"frame_interval_s": "(float or None) - Seconds between consecutive timepoints, obtained from the acquisition metadata. It converts frame indices into physical time in the tracks table and displacement per frame into speed. It does not affect object linking, so an incorrect value rescales reported velocities without changing track identities. None uses the motility module's seconds_per_frame setting. Default None.",
"t_track_backend": "(str) - Linker used to join objects between consecutive timepoints. 'iou' compares complete object volumes and requires no distance or anisotropy parameter, but cannot link an object that moves farther than its own width between frames. 'centroid' links nearest centroids within the displacement limit and supports faster movement, but requires an appropriate limit. Use 'iou' for crowded fields with slow motion and 'centroid' for sparse fields with rapid motion. Default 'iou'.",
"t_link_threshold": "(float) - Minimum intersection-over-union, between 0 and 1, required to assign an object at one timepoint to the same track at the next when t_track_backend is 'iou'. Increasing it can divide moving or growing objects into short tracks; decreasing it can join neighbouring objects whose volumes overlap. This parameter is separate from stitch_threshold because overlap distributions differ between consecutive z planes and consecutive timepoints. Matching is one-to-one. Default 0.25.",
"t_max_displacement_px": '(float or None) - How far an object may move between consecutive timepoints and still count as the same object, in image pixels, for the distance-based backends. The z component is multiplied by anisotropy first, so a one-plane move on a stack with a 5x z step costs 5 px of the budget rather than 1. Too small breaks tracks at every fast frame; too large joins neighbours into one. Default None.',
"t_max_displacement_um": "(float or None) - Maximum between-frame movement, equivalent to t_max_displacement_px but expressed in micrometres. Conversion requires voxel_size_z_um and voxel_size_xy_um; with both values defined, anisotropy is incorporated into the physical coordinates. Set either this value or t_max_displacement_px, not both, because they define the same displacement threshold in different units. Default None.",
"t_project_for_tracking": "(bool) - Collapse each timepoint's z-stack to one plane before linking, so tracking uses the projection while segmentation uses the volume. Enable this setting when volumetric linking is too slow or anisotropy is uncertain. Objects at the same lateral position but different z positions then merge in the projection and cannot be distinguished downstream. This setting does not enable backends that do not support volumetric data. Default False.",
"save_original_images": "(bool) - After each batch is MIP-projected and merged into stack/, either move the raw input images into src/orig/ (True) or delete them so the pixels live only in stack/ (False). Set False on large screens where the duplicate raw copy will not fit on disk; the deletion is not reversible. Default True.",
"keep_intermediate": "(bool) - Keep the intermediate stack/ and masks/ folders after the merged/ arrays are built. Off by default: only merged/ is kept (masks are embedded in merged and recorded in the database).",
"mask_parallel": "(bool) - Segment the prepared cell, nucleus and pathogen batches on several GPUs at once, one model process per GPU. Each batch goes to exactly one GPU, finished batches are kept and a rerun with the same settings resumes the rest. Masks match the single-GPU run. Needs two or more CUDA or ROCm GPUs and the Cellpose backend; not for timelapse or t_stack runs. With adjust_cells, adjusted cells are written to masks/adjusted_cell_mask_stack and the raw cell masks are kept. Default False.",
"mask_gpu_indices": "(str) - GPUs used when mask_parallel is on, as comma-separated numbers such as 0,1. Blank uses every GPU the process can see, which on a cluster means the GPUs allocated to the job; with fewer than two the run uses one device. Default blank.",
"watch_folder": "(bool) - Watch src and analyse each field as its images arrive. A file must stop changing for watch_settle_seconds and read whole. A fixed conversion_map.csv requires every declared companion. Without a map, known numeric filename conventions require every channel from the documented first ID through the highest selected position; custom or named channels need a fixed map. Each field runs alone, like batch_size 1, and results collect in src/spacr_watch; its ledger skips completed fields on restart. Not for timelapse, z_stack or t_stack. Default False.",
"watch_pipeline": "(str) - What watch_folder runs on each arriving field. 'mask' runs Make Masks; 'mask_measure' adds Measure; 'mask_measure_classify' also applies a saved CV model to measured objects. Results gather in src/spacr_watch/measurements/measurements.db. Default 'mask'.",
"watch_normalization_pool": "(str) - 'per_field' analyses arriving fields independently. 'fixed_map' waits for every image declared in conversion_map.csv, then runs the ordinary static projected v1 Mask or Mask/Measure batch pipeline with shared legacy percentile normalization and padding. Requires randomize False and batch_size greater than 1. Incomplete cohorts remain waiting; timeouts never release partial pools. Membership, recipes and original input hashes bind one restart-safe checkpoint. Default 'per_field'.",
"watch_measure_settings": "(str) - A Measure settings file (.csv or .json, as the Measure screen saves them) used by watch_pipeline 'mask_measure' and 'mask_measure_classify' for every field. Blank uses Measure's defaults with this run's channels. Default blank.",
"watch_classify_settings": "(str) - A saved Classify settings file for the mask_measure_classify watch pipeline. It must select CV inference from an existing model_path, with crop_source merged, apply_model_to_dataset on, and train, test and generate_training_dataset off. The model is copied once for the watch, and predicted classes and scores join the combined measurements database. Default blank.",
"watch_settle_seconds": "(float) - How long an image must keep the same size and modification time before watch_folder reads it, so a file the microscope is still writing is not taken half-written. Raise it for slow network shares. Default 10.",
"watch_poll_seconds": "(float) - How often watch_folder looks in src for new or changed images. A field is picked up about watch_settle_seconds plus this long after its last file stops changing, once the fields before it are done. Default 5.",
"watch_idle_minutes": "(float) - Stop watching once nothing in src has changed for this many minutes, and list the fields that never became complete. 0 watches until Stop is pressed. Default 0.",
"microscope_feedback": "(bool) - With a measured watch pipeline, send objects matching microscope_event_query back to the microscope for re-imaging, for example at higher resolution or as a short timelapse. Stage positions come from microscope_positions and microscope_stage_transform; every event and its images are recorded in watch_ledger.json and the images saved in src/spacr_watch/reimaged. Default False.",
"microscope_driver": "(str) - The microscope microscope_feedback drives. 'simulated' acquires from the images in microscope_simulated_folder laid out at microscope_positions, for trying the loop without a microscope. 'pycromanager' drives a running Micro-Manager through pycro-manager (pip install pycromanager, and turn on Micro-Manager's server under Tools > Options). Default 'simulated'.",
"microscope_simulated_folder": "(str) - The folder of field images the simulated microscope acquires from, named as the watched images are and placed on the stage by microscope_positions. Blank uses src. Default blank.",
"microscope_positions": "(str) - A table (.csv, .xlsx or .parquet) of the stage position each field was acquired at, with the columns field (the field name the watch prints, such as plate1_A01_0001_001), x and y in micrometres at the image centre, and optionally z. Events in a field missing from it are recorded but not imaged. Default blank.",
"microscope_stage_transform": "(list) - Four numbers [a, b, c, d] turning a pixel offset from the image centre (dx columns, dy rows) into a stage move of (a*dx + b*dy, c*dx + d*dy) micrometres. For a pixel size p with camera and stage axes aligned use [p, 0, 0, p]; negate a term for a flipped axis and swap them for a camera turned 90 degrees. Default [1.0, 0.0, 0.0, 1.0].",
"microscope_event_table": "(str) - The measurement table whose objects can become events, such as cell, nucleus or pathogen. Each event is placed at the object's intensity-weighted centroid. Default 'cell'.",
"microscope_event_query": "(str) - A pandas query on microscope_event_table choosing which objects are events, for example pathogen_area > 200 or cell_channel_1_mean_intensity > 900. Blank makes every object an event, up to microscope_max_events per field. Default blank.",
"microscope_max_events": "(int) - The most events one field sends to the microscope, taken in table order, so a field full of matches does not hold up the plate. Default 10.",
"microscope_timepoints": "(int) - How many images the microscope takes at each event: 1 is a single snapshot, more makes a timelapse spaced microscope_interval_seconds apart. The microscope keeps its current channel, objective and exposure. Default 1.",
"microscope_interval_seconds": "(float) - Seconds between the timelapse images of one event when microscope_timepoints is above 1. The watch waits at the event meanwhile, so long timelapses delay later fields. Default 0.",
"cloud_anonymous": "(bool) - Read cloud sources without credentials, for public data such as IDR or the Cell Painting Gallery. When off, s3:// sources use the AWS credentials in the AWS_* environment variables, ~/.aws or AWS_PROFILE, and are read anonymously only when none are found; gs:// and az:// use their own default credentials. spaCR never stores or prints credentials. Default False.",
"cloud_profile": "(str) - Named AWS profile from ~/.aws/config used for s3:// sources, for example lab-readonly. Only the name is kept; the keys stay in ~/.aws. Blank uses the default AWS credential chain. Default blank.",
"cloud_endpoint": "(str) - Address of the S3-compatible service holding s3:// sources, such as a MinIO or Ceph server, or https://uk1s3.embassy.ebi.ac.uk for IDR. Blank uses Amazon S3. Default blank.",
"cloud_cache": "(str) - Folder where cloud sources are fetched to and analysed in. Each src address gets its own subfolder, so a repeated run reuses what was fetched and Measure finds what Make Masks wrote. Blank uses ~/.cache/spacr/cloud. Default blank.",
"cloud_wells": "(str) - Wells of a cloud OME-Zarr plate to fetch, such as A1, B03; only these are downloaded, a channel at a time, with z maximum projected. Blank fetches every well, which for a large plate can be many gigabytes. Default blank.",
"cloud_fields": "(int) - Fields per well of a cloud OME-Zarr plate to fetch, counted from the first. 0 fetches every field. Default 0.",
"cloud_level": "(int) - Resolution level of a cloud OME-Zarr to fetch. 0 is full resolution; each level above usually halves width and height, so set diameters and size limits for the smaller images. Default 0.",
"cloud_results": "(str) - Cloud folder (s3://, gs:// or az://) the run's measurements folder is copied to when it finishes, in one subfolder per src. Blank keeps results in the local folder only. Default blank.",
"keep_original_images": "(bool) - Keep the original raw input images (in orig/). Off by default to save disk space; the pixel data lives in merged/.",
"amsgrad": "(bool) - Use the AMSGrad variant of Adam/AdamW, which keeps a running maximum of past squared gradients instead of their decaying average so the effective step size never grows back. Enable when training loss oscillates or stops converging with plain Adam; it costs a little speed and memory. Only honoured by optimizer_type 'adam' and 'adamw' - ignored by sgd, rmsprop, nadam, radam and adagrad. Default True.",
"analyze_clusters": "(bool) - After clustering the embedding, rank every measured feature by cluster separation using random-forest importance and a per-feature ANOVA or Kruskal-Wallis test, then write results/cluster_results.csv. Enable this setting to identify morphology or intensity features associated with each cluster. It adds a full model fit over the feature table. Default False.",
"augment": "(bool) - Expand the training split eightfold by adding four 90-degree rotations of each crop and their horizontal reflections; validation and test splits are not augmented. Enable this setting when few annotated objects are available and validation accuracy is below training accuracy. The expanded set is materialised in RAM, requiring approximately eight times the memory and epoch duration. Default False.",
"background": "(float) - Per-channel background level in raw intensity units. Pixels below it are zeroed when remove_background is on, and it is multiplied by Signal_to_noise to set the upper anchor for normalization. Raise it if faint haze survives; set it too high and dim real objects vanish. Default 100 (200 for Cellpose training and plaque analysis).",
"backgrounds": "(list of float) - Legacy compatibility field retained in settings snapshots. Current mask preprocessing ignores this list and reads cell_background, nucleus_background, pathogen_background and each organelle background setting instead, so changing it does not alter segmentation. Default [100, 100, 100, 100].",
"black_background": "(bool) - Choose the standalone/CLI embedding fallback: black canvas with white axes when True, white canvas with black axes when False. In the Qt app, Image UMAP automatically matches its enclosing card in the active theme and uses that theme's readable foreground color instead. Default True.",
"calculate_correlation": "(bool) - For every pair of measured channels and every object mask, compute a per-object Pearson correlation and the three Manders coefficients (manders_m1, manders_m2, manders_overlap_coefficient), stored as <object>_channel_i_channel_j_* columns. Needs at least two channels. Turn it off to cut measurement time and database size when colocalisation is not part of the phenotype. Default True.",
"cell_background": "(int) - Background intensity of the cell channel in raw image units. Pixels below it are zeroed when remove_background_cell is True, and it is multiplied by cell_signal_to_noise to set the intensity the normalisation ceiling must reach. Set it from a genuinely empty region; too high and dim cells are erased. Default 100.",
"nucleus_background": "(int) - Raw intensity value treated as background in the nucleus channel. When remove_background_nucleus is True, every pixel below it is zeroed before normalization; it is also multiplied by nucleus_signal_to_noise to set the upper-clip target. Raise it for images with high offset or autofluorescence, lower it if dim nuclei disappear. Default 100.",
"pathogen_background": "(int) - Assumed background intensity of the pathogen channel in raw image units. It has two jobs: when remove_background_pathogen is True every pixel below it is zeroed, and it is multiplied by pathogen_signal_to_noise to set the brightness the normalisation ceiling must reach. Raise it if dim haze is being segmented; lower it if faint parasites vanish. Default 200.",
"cell_chann_dim": "(int) - Recruitment analysis only (analyze_recruitment): the image-channel index paired with the cell mask when drawing outline overlays, and the switch that enables the cell filters - set an integer and cell_size_range, cell_intensity_range and target_intensity_min are applied; leave it None and cells are not filtered at all. Default 3.",
"cell_intensity_range": "(list) - Legacy [min, max] bounds used during recruitment analysis when cell_chann_dim is set. Despite the setting name, the current _object_filter call uses index 0 from [nucleus, pathogen, cell] and therefore filters the nucleus-channel mean intensity. Review the filtered object counts when using this setting. Default None.",
"cell_loc": "(list) - One list of well identifiers per entry in cells, specifying the plate locations of each host cell line, for example [['c1','c2'],['c3']]. Identifiers must start with 'r' for a row or 'c' for a column; other values are ignored and the corresponding wells remain unannotated. None labels every row with the first entry of cells. No default is set: annotate_filter_vision accesses settings['cell_loc'] directly, so the key must be present in the dictionary and explicitly set to None when location mapping is not required.",
"cell_mask_dim": "(int) - Position along the last axis of each merged/*.npy array where the cell label mask sits. Merged arrays are ordered [image channels..., cell, nucleus, pathogen, organelle], so the default 4 assumes the four channels 0-3 were kept; keep fewer channels and every mask dim shifts down. None makes measure_crop skip all cell measurements and cell crops. Default 4.",
"cell_min_size": "(int) - (Deprecated) Pixel-area floor applied to cell labels during measurement: any cell smaller than this is erased from the mask before features are extracted. Superseded by an 'area' row for cell in object_filters, which filters at segmentation time, but this one still runs if you set it. 0 or None disables it. Default 8000.",
"cell_plate_metadata": "(list of lists) - Wells occupied by each entry of cell_types, with one inner list per cell type in the same order, for example [['c2','c3'],['c4']]. Each identifier must start with 'c' (column) or 'r' (row); invalid identifiers are skipped without an exception and those wells receive no host_cells label. Because 'condition' combines the labels that are present, a typographical error changes the comparison without raising an error. Default None.",
"cell_signal_to_noise": "(int) - Multiplied by cell_background to define the minimum intensity for the normalisation ceiling. spaCR evaluates the 98th through 99.5th percentiles of the cell channel and uses the first value at or above that product as the upper anchor. Increase it to raise the ceiling and reduce normalised intensity; decrease it to increase the visibility of faint cells. Default 10.",
"cell_size_range": "(list) - [min, max] bounds in pixels^2 on cell_area, used to drop rows from the measurement table during recruitment analysis; only cells strictly between the two values are kept. Both entries must be integers or that bound is silently skipped. Setting it to None widens it to [0, 1e100]. Default [0, 100000].",
"cell_types": "(list) - Names of the host cell lines in the experiment, e.g. ['HeLa']. Each name is written into the host_cells column and folded into the combined condition label used for grouping and plotting; the list is positionally paired with cell_plate_metadata, which says which wells hold each one. Default ['HeLa'].",
"cells": "(list) - Names of the host cell lines on the plate, e.g. ['HeLa']. Each name is written to the host_cells column and becomes part of the combined condition label used for grouping in plots and statistics. With cell_loc set the names are mapped well by well; with cell_loc None only the first name is used, applied to every row. No default is set: no set_default_* function fills this key and its readers index settings['cells'] directly, so it must be present - use None to skip host-cell annotation.",
"cells_per_well": "(int) - Minimum cells a well must contribute to survive recruitment analysis; wells below it, and every cell in them, are dropped before the by-well plots and CSVs are produced. Raise it to suppress noisy, sparsely populated wells at the cost of losing those wells. Default 0, which keeps every well.",
"channel_dims": '(list) - Recruitment analysis only: image-channel indices in the merged arrays. They determine the channels used for overlays and recruitment measurements, each auxiliary recruitment ratio records its channel explicitly, so measurements from different channels remain separate. Default [0, 1, 2, 3].',
"channel_of_interest": "(int, list, or str) - Measurements available to the model. Specify one channel to train on that channel alone, multiple channels to use their combination of measurements, or 'shape' to use outline-derived features. An empty value includes every measurement. A colocalisation feature is associated with both measured channels, so selecting either channel also includes their shared colocalisation features. This setting also selects the channel used for recruitment measurements. Default 3 in machine-learning steps and 1 or 2 elsewhere.",
"chunk_size": "(int) - Number of FASTQ reads read into memory and handed to each worker batch. Larger chunks cut per-batch overhead and make the progress bar coarser but raise peak RAM per job; smaller chunks stream more gently on low-memory machines. Also sets how many reads are processed when test is True. Default 100000.",
"classes": "(dict) - Class definitions in the form class name -> {column, value}; for example, 'pc' might be {'column': 'columnID', 'value': 'c3'}. In the Classes editor, select a column and assign names to its distinct values; definitions may span several columns. One row may instead define a random complement containing objects not assigned by any other class, sampled to match the largest class. This setting defines which objects belong to each class; class_folder_names defines the training subfolders. A pre-split settings file contains a plain list and is converted when read. Default {}.",
"class_folder_names": "(list of str) - Ordered training folder names. Each must exactly match a subfolder under src/train and src/test; its position becomes the integer label, and the list length sets the classifier-head width. Training raises FileNotFoundError with missing and available folders when a name is absent. Generate Training Dataset replaces this list with the folders it wrote. This setting identifies crop locations; 'classes' defines their semantic labels. Default ['nc','pc'].",
"clustering": "(str) - Algorithm applied to the two-dimensional embedding. 'dbscan' identifies density-based clusters from eps and min_samples, labels sparse points as noise (-1), and determines the cluster count from the data. 'kmeans' forces exactly min_samples clusters and assigns every point. Use dbscan for distinct phenotypes over a diffuse background and kmeans when a fixed number of groups is required. Default 'dbscan'.",
"col_to_compare": "(str) - Metadata column that identifies the control wells when embedding_by_controls is True: rows whose value equals pos or neg are used to train the reducer, and the column is then dropped before fitting. Typically 'columnID' or 'rowID' depending on where controls sit on the plate. Ignored otherwise. Default 'columnID'.",
"color_by": "(str) - Name of a column in the joined measurement table (e.g. 'cond', 'columnID', 'plateID') used to color embedding points instead of the cluster labels. Set it to see how a known grouping such as condition or plate column falls across the map; leave it None to color by the clustering result. Setting it also disables remove_cluster_noise, plot_outlines and smooth_lines. Default None.",
"consolidate": "(bool) - Before processing, recursively scan src for images and copy them into a single <src>/consolidated folder, prefixing each filename with its subfolder names so nothing collides; src is then repointed there. Use it when one plate's images are split across per-well or per-channel subfolders. Copies, so disk use roughly doubles. Default False.",
"CP_prob": "(float) - Cellpose cellprob_threshold: the cell-probability cut-off applied to the network output when deciding which pixels belong to an object. Lower it (typically toward -6) to recover dim or partly detected objects and grow existing masks; raise it (toward 6) to drop faint false positives and shrink masks. Default 0.",
"custom_model": '(str) - Path to a saved Cellpose model, loaded as pretrained_model by the mask-finetune tool. When set, model_type is passed as None and diameter as diam_mean (which Cellpose 4.x ignores with a warning). model_name remains active and selects the channel pair sent to model.eval; an incompatible value therefore segments the wrong channels. Default None.',
"cytoplasm_min_size": "(int) - (Deprecated) Pixel-area floor for the cytoplasm mask, which is the cell mask with nucleus, pathogen and organelle pixels removed. Cytoplasm regions below this are erased before measurement, so their host cell yields no cytoplasm features and any recruitment ratio built on them is lost. 0 or None disables. Default 0.",
"nucleus_min_size": "(int) - (Deprecated) Minimum nucleus size in pixels^2 applied during measure_crop: labels covering fewer pixels than this are erased from the nucleus mask before any feature is measured, so those nuclei never reach the database. 0 (default) disables it. Prefer an 'area' row for nucleus in object_filters, which filters at segmentation time.",
"dependent_variable": "(str) - Name of the column in score_data that is modelled as the response, e.g. 'pred'/'predictions' from the ML scoring step or a measured feature such as 'pathogen_nucleus_shortest_distance'. It is aggregated per well by agg_type and then optionally transformed. The run aborts if the column is absent from the score CSV. Default 'pred'.",
"score_column": "(str) - Which column of the prediction CSV holds the CNN score that Explain CV and the hit-investigation montages read. The regression module no longer has this setting: it fits dependent_variable and simulates the minimum cell count on that same column, so one measurement cannot be named two ways there. Default 'cv_predictions'. Investigate Hit starts blank because no score field can be inferred universally; select the prediction column before building its montages.",
"analysis_mode": "(str) - 'regression' fits the selected simultaneous model. 'guide_permutation' tests each guide as a plate-adjusted marginal association using blocked Freedman--Lane permutations and then corrects the requested support family. This setting is normally derived from inference; set it directly only to override that choice. Default 'regression'. The Regression module starts with inference='nonparametric', so its resolved initial mode is 'guide_permutation'.",
"inference": "(str) - How effects are tested; the readable front end for analysis_mode. Default 'nonparametric': each guide is a plate-blocked Freedman-Lane permutation with an empirical P, valid however many guides there are, but no P can be below 1/(guide_permutations + 1). 'parametric' fits every guide at once in the chosen regression_type, so it needs more wells than guides or no coefficient is identifiable. 'auto' counts guides and wells and takes the simultaneous fit only when the design supports it.",
"analysis_unit": "(str) - What one row of the model is. 'well' collapses each well's objects into a single value with agg_type first, so the well is the independent unit and the number of cells behind it only affects precision. 'cell' regresses the individual objects instead, which keeps power but treats cells from one well as independent when they are not, so standard errors are optimistic unless the model accounts for the clustering (regression_type='mixed'). This is the explicit spelling of agg_type=None, which used to change the unit of analysis silently. Default 'well'.",
"guide_min_wells": "(int or list) - Minimum numbers of independent wells containing a guide. A list such as [1, 2, 3, 4] writes one sensitivity-analysis table and volcano plot per threshold; P values are computed once and the multiple-testing correction is repeated within each eligible family. Default [1, 2, 3, 4].",
"guide_primary_min_wells": "(int or None) - Which guide_min_wells family supplies results_significant.csv and the returned 'significant' table. Default None chooses the smallest requested threshold.",
"guide_permutations": "(int) - Number of plate-blocked Freedman--Lane residual permutations used for empirical two-sided guide P values. The estimator is (exceedances + 1) / (permutations + 1), where exceedances are permuted statistics at least as extreme as the observed statistic; this imposes a hard floor of 1 / (permutations + 1). Values of 1,000, 10,000, and 200,000 resolve floors of approximately 1e-3, 1e-4, and 5e-6, respectively. Increase the count when results accumulate at this resolution floor; runtime increases linearly. Default 200000.",
"guide_permutation_seed": "(int) - Random seed for reproducible residual permutations. Keep it fixed to reproduce exact empirical P values; change it to check Monte Carlo sensitivity. Default 0.",
"grna_statistic": "(str) - What the permutation test measures between a gRNA's well fractions and the well phenotype. 'pearson' is a partial correlation, which is linear and is moved by an extreme well in proportion to how extreme it is. 'rank' is the same quantity computed on the ranked phenotype, so it responds to order rather than magnitude and no single well can move it far. Both cost one matrix product, so the choice does not change how long the test takes. Default 'pearson'.",
"guide_permutation_block": "(str) - Column defining exchangeability blocks for permutations, normally plateID. Residuals are never shuffled between its levels. Default 'plateID'.",
"guide_nuisance_columns": "(list) - Additional measured well-level covariates to residualize from both phenotype and guide fraction before testing. Do not put post-treatment outcomes here. Default []. Regression starts with ['rowID', 'columnID'] to remove plate-position structure before the within-plate permutation.",
"guide_presence_threshold": "(float) - A guide counts as present in a well only when its fraction is above this value. The effect still uses the unthresholded fraction. Default 0.0.",
"guide_permutation_batch_size": "(int) - Number of permutation outcomes evaluated together. Lower this if memory is tight; it does not change the result. Default 500.",
"multiple_testing_method": "(str) - Correction applied within each outcome/support family: fdr_bh (Benjamini--Hochberg, default), fdr_by, bonferroni, holm, or none. Stricter family-wise methods generally call fewer guides.",
"p_threshold_alpha": "(float) - P-value threshold used to call a coefficient a hit and to draw the volcano-plot reference line. It applies to the P-value type selected by p_threshold_kind, keeping results_significant.csv and the corresponding figure consistent. Supply a fraction strictly between 0 and 1; 5 is rejected as an invalid representation of 5%. Default 0.05.",
"p_threshold_kind": "(str) - Whether p_threshold_alpha is applied to the multiple-testing-adjusted p-value ('adjusted') or the raw per-coefficient p-value ('raw'). The analysis run and volcano plot use the same choice, so exported hits and plotted calls remain consistent. 'raw' generally calls more genes in screens with thousands of guides. Other values are rejected. Default 'adjusted'.",
"rra_alpha": "(float) - The top fraction of the ranked guide list robust rank aggregation scores against: 0.25 asks whether a gene's guides cluster in the best quarter of the ranking more than chance allows, ignoring the rest. Smaller is stricter and returns fewer, better-supported genes. Above 0 and at most 1; 25 for '25%' is refused. Default 0.25.",
"rra_permutations": "(int) - How many permuted rankings the robust rank aggregation null is built from. The smallest P value it can report is about 1/rra_permutations, so 10000 resolves the tail to 1e-4; raise it when many genes pile up at that floor and lower it while exploring, since the cost is linear in this number. Default 10000.",
"count_grna_column": "(str) - Name of the column in the count CSV containing the guide identifier. Earlier versions required the name 'grna' and rejected files using alternatives such as 'sgRNA' or 'guide'. Set this value to the column produced by the sequencing pipeline. Default 'grna'.",
"count_value_column": "(str) - Name of the column in the count CSV holding the read count for one guide in one well; it becomes the per-well fraction the fraction_threshold sweep works on. Hard-coded to 'count' until now, so a file naming it 'reads' or 'n' failed with a message naming only the columns spaCR expected. Default 'count'.",
"independent_variable_layout": "(str) - Shape of the independent-variable/count table: 'long' means one row per well and guide, 'wide' means one row per well with one guide per column, and 'auto' detects long from count_grna_column plus count_value_column and otherwise treats the numeric non-metadata columns as guides. Wide input is melted losslessly before filtering. Default 'auto'.",
"wide_predictor_columns": "(list) - Guide columns in a wide independent-variable table. Leave empty to use all numeric columns other than plate/well metadata; list them explicitly when the table contains additional numeric metadata. Ignored for long input. Default [].",
"model_data_layout": "(str) - Shape handed to a fixed-effects estimator. 'long' preserves the historical repeated well-guide formula; 'wide' pivots guide or gene fractions to one row per independent well before fitting. Mixed models and Freedman-Lane permutation testing require the long representation and convert wide input back to long automatically. Default 'long'.",
"fdr_alpha": "(float) - Family-level rejection threshold for adjusted P values in guide_permutation mode. Must be between 0 and 1. Default 0.05.",
"tolerance": "(int or float) - How close a subsampled well mean has to be to the full-well mean before minimum_cell_simulation calls that sample size sufficient, which is what sets min_cells_per_well when you leave it None. An int is read as a percentage (2 means 2%), a float as a fraction (0.02 means the same); anything else raises ValueError. Tighten it toward 0.01 to demand more cells per well and drop more wells, loosen it to 0.05 to keep sparse wells at the cost of noisier per-well scores. Default 0.02.",
"invert_dependent_variable": "(bool or int) - Transform the response before per-well aggregation when lower scores represent a stronger phenotype. False or 0 leaves the response unchanged, True or 1 uses 1 - x (appropriate for probabilities), and -1 uses 1 / x (appropriate for distances or counts). Any other value raises ValueError in process_scores. The transformation changes coefficient signs and therefore the side of the volcano plot on which significant effects appear. Default False.",
"y_lims": "(list or None) - Limits of the -log10(p) axis of the Toxoplasma volcano plot. None auto-scales to the data; [low, high] fixes the axis so several plates can be compared at the same scale; [[low1, high1], [low2, high2]] draws a broken axis with the gap between the two ranges removed, which keeps a handful of extremely significant genes on the plot without flattening everything else. Any other shape raises ValueError. Default None.",
"dialate_png_ratios": "(list of float) - Dilation amount as a fraction of object size: the mask is grown by ratio * sqrt(object area) pixels of binary dilation, so 0.2 expands a cell by roughly 20% of its diameter and pulls in surrounding background. Only used when dialate_pngs is True. A single value applies to every crop_mode entry; pass a list only when the modes need different ratios. Default [0.2].",
"dialate_pngs": "(bool) - Grow each object mask before cropping so the PNG keeps a rim of surrounding pixels instead of a hard mask edge; the amount comes from dialate_png_ratios. May be a list with one value per crop_mode entry (a single value applies to all of them), and is forced off for crop_mode 'cytoplasm'. Enable when context around the object helps the classifier. Default False.",
"dot_size": "(int) - Matplotlib marker area, in points squared, for each object plotted in the UMAP/tSNE embedding. Increase it when a few hundred points make the scatter look empty; drop it to roughly 5-10 when tens of thousands of points overplot and hide cluster structure. Default 50.",
"point_color": "(str) - Point color for static and interactive UMAP plots. Use 'cluster' or 'viridis' for cluster-based Viridis colors, or any Matplotlib color such as '#4cc9f0', 'orange', or 'white' for one fixed color. Default 'cluster'.",
"point_alpha": "(float) - Opacity of UMAP points from 0 (invisible) to 1 (opaque), used by both static and interactive plots. Default 0.65.",
"outline_width": "(float) - Width in points of cluster outlines and interactive selection rings. Smaller values produce thinner boundaries. Default 1.0.",
"umap_canvas_width": "(int) - Initial interactive UMAP chart width in pixels. The chart/sidebar divider can also be dragged while exploring. Default 900.",
"umap_sidebar_width": "(int) - Initial interactive UMAP image and annotation sidebar width in pixels. The divider remains draggable. Default 280.",
"dropout_rate": "(float) - Dropout probability (0-1) written into every existing Dropout layer of the backbone and applied to a Dropout inserted before the final linear classifier; 0 or None removes dropout entirely. Raise it (0.2-0.5) when training accuracy runs well ahead of validation accuracy; lower it when the model underfits and training loss stalls high. Default 0.1.",
"eps": "(float) - DBSCAN neighbourhood radius, expressed in the units of the UMAP/t-SNE embedding and measured with the 'metric' setting: two points are neighbours if they lie within this distance. Raise it to merge fragments into fewer, larger clusters and leave less noise; lower it to split clusters and push more points to noise (-1). Ignored when clustering is 'kmeans'. Default 0.9.",
"epochs": "(int) - Number of full passes over the training set. It also sets the learning-rate schedule horizon - cosine anneals over exactly this many epochs and step_lr drops every epochs/5 - so changing it rescales the schedule. A checkpoint is always written on the final epoch and every 100th. Raise it for small datasets and use early_stopping_patience to cut runs short. Default 100.",
"examples_to_plot": "(int) - How many randomly chosen merged image stacks are rendered as segmentation-overlay previews after mask generation (in timelapse mode, per-channel panels instead). Raise it to check outlines and normalization across more fields of view, at the cost of render time and larger PDFs; 0 skips previews entirely. Default 1.",
"positive_control_wells": "(list or str) - Wells containing only the positive control, e.g. ['c2']. Accepts rows (r1), columns (c1), or individual wells (A01); the Plate button selects them from a map. Pure controls are calibration references rather than screen observations, so they are excluded from the regression. They define the positive endpoint for mixed-ratio calibration and should be identified from the plate design. Default None.",
"negative_control_wells": "(list or str) - Wells containing only the negative control, e.g. ['c1']. Accepts the same row, column, and well notation as positive_control_wells. These wells are excluded from the regression and define the negative endpoint for mixed-ratio calibration. Default None.",
"mixed_control_wells": "(list or str) - Wells holding a known mixture of the positive and negative controls, e.g. ['c3']. Removed from the regression like the other two. These provide strong validation because per-cell identities are unknown while the aggregate proportions are known from sequencing, allowing annotation methods to be scored on real rather than simulated cells. Default None.",
"exclude_grnas": "(list or str) - gRNA or gene identifiers known not to occur in cells, such as primer or plasmid carry-over. These sequences are removed before guide fractions are calculated, and retained guides are renormalized. This differs from background subtraction, which corrects spurious reads assigned to a real guide. A gene identifier matches all of its guides. Default None.",
"exclude": "(str or list) - Names of measurement columns to drop from the feature set before UMAP embedding or ML training, applied after the channel_of_interest selection. Use it to remove features that leak the label or swamp the embedding. It does not filter database rows; use exclude_rows for that. Default None keeps every feature.",
"exclude_conditions": "(list) - Condition labels dropped from the image UMAP input, matched against the cond column that map_condition derives from the pos, neg and mix column IDs; the only possible entries are 'neg', 'pos', 'mix' and 'screen'. A bare string is accepted and wrapped in a list. Use it to embed screen wells only. Default None.",
"exclude_rows": "(dict or None) - General UMAP row exclusions. Choose one or more database columns, then check the values whose rows should be removed. Rules are combined with OR, so a row matching any selected column/value pair is excluded. Default None keeps every row.",
"experiment": "(str) - Free-text run label. Its real effect is naming the exported PNG dataset tar as <YYMMDD>_<experiment>.tar (a random-numbered variant is used if that name already exists), so give each screen a distinct value to avoid confusing dataset tars. It is also passed to the measurement-database writer but not stored there. Default 'experiment' (the barcode pipeline uses 'experiment_1' and a foreign import uses 'foreign_import').",
"figuresize": "(int) - Base figure size in inches; figures are built square as figuresize x figuresize and font sizes are derived from it (legend, axis labels and ticks at 0.75x, overlay text at 0.5x). Raise it when text is unreadable at publication scale, lower it to fit panels on screen. Default 10; cluster grids cap total width at 200 inches.",
"filter_by": "(str or None) - Restricts the feature matrix before dimensionality reduction: only columns matching this channel are kept and the other channel_1-channel_4 columns are dropped. Accepts 'channel_0'-'channel_3', an int, a list of channel numbers, or 'morphology' to keep only shape features (area, eccentricity, Zernike moments, ...). None, 'None', 'all', and '*' disable filtering. Default 'channel_0'.",
"fill_in": '(bool) - Fill the holes inside each object of every mask Cellpose Masks (Apply) writes. Each object is filled on its own and keeps its id, so touching objects stay separate and a hole never takes pixels from a neighbour. Default True in Cellpose Masks.',
"flow_threshold": "(float) - Cellpose flow_threshold: the maximum allowed error between the predicted flow field and the flows recomputed from each candidate mask; masks above it are discarded. Raise it to keep more objects, including irregularly shaped ones; lower it to reject poorly formed masks and reduce false positives. Default 0.4.",
"fps": "(int) - Playback rate of the per-channel movies written to <src>/movies from timelapse .npy stacks, and only when timelapse is True. Raise it to skim long acquisitions, lower it to inspect individual frames. Affects the movies only - never tracking, segmentation or measurements. Default 2.",
"calibrate_fraction_threshold": "(bool) - Estimate fraction_threshold from control wells instead of using the configured value. The sweep recomputes per-well fractions across candidate cutoffs and selects the cutoff with greatest imaging-sequencing agreement. The plate design must identify pure control wells independently; selecting controls by the measured fraction would be circular. Default False.",
"fraction_threshold": "(float or None) - Minimum relative abundance, 0-1, that a gRNA must reach within a well's total read count to be retained. Increasing it removes low-abundance and bleed-through gRNAs and reduces the mean number of gRNAs per well; if set too high, every row is removed and the run raises an error. Use None to select automatically the cutoff that yields target_unique_count gRNAs per well. Default None. Regression starts at the reproducible fixed cutoff 0.02; enter None only when the automatic sweep is intended.",
"normalise_fraction": "(bool) - Divide a gRNA's fraction by the sum of the fractions that remain in its well after fraction_threshold, before deciding how many cells it is given. On, a gRNA's share is measured against what survived the threshold; off, it is measured against every read the well produced, including those the threshold removed. The two differ whenever the threshold removes anything: normalising raises every surviving share, and by more the more was removed. Default True.",
"from_scratch": "(bool) - Start from randomly initialised weights instead of fine-tuning the pretrained model. Keep this disabled unless the training set contains enough images to fit a model without pretrained weights; fine-tuning may require tens of images, whereas training from scratch commonly requires thousands. Default False.",
"mixed_precision": "(bool) - Enable automatic mixed precision for supported accelerator operations during training. On a modern graphics card it can make training roughly twice as fast while using less memory, although the gain depends on the model and hardware. Floating-point differences can affect exact reproducibility, so compare runs only when they use the same precision mode. Unsupported hardware disables this mode with an explicit message. Default False.",
"gradient_accumulation": "(bool) - Sum gradients over several batches before each optimizer step, producing an effective batch size of batch_size x gradient_accumulation_steps without additional GPU memory. Enable when batch_size must be reduced to fit available VRAM and the resulting training trajectory is unstable. Remaining gradients are applied at the end of each epoch. Default True.",
"gradient_accumulation_steps": "(int) - How many batches are summed per optimizer step when gradient_accumulation is on; the loss is divided by this value so gradient magnitude stays comparable. Effective batch size = batch_size x this. Raise it (4-16) to emulate a larger batch on limited VRAM, at the cost of fewer weight updates per epoch. Ignored when gradient_accumulation is False. Default 4.",
"grayscale": "(bool) - Force the Cellpose channel pair to [0, 0] so the network treats the input as a single combined channel, overriding the [cytoplasm, nucleus] pair otherwise inferred from model_name (cyto -> [1,0], cyto2 -> [2,1], nucleus -> [0,0]). Leave it on for single-channel inputs; switch it off only when feeding a genuine two-channel stack. Default True.",
"grouping": "(str) - How per-object values collapse to one number per well in the plate heatmap: 'mean' averages heatmap_feature over the objects in a well, 'sum' totals them, 'count' ignores the feature and colors wells by object count. Use 'count' to spot uneven seeding or dropout, 'mean' for phenotype strength. Default 'mean'; any other value raises ValueError.",
"heatmap_feature": "(str) - Numeric column that is aggregated per well and color-mapped in the plate heatmap after ML scoring, e.g. 'predictions' for the classifier score or 'recruitment' for the pathogen/cytoplasm intensity ratio. Must be a numeric column of the scored dataframe or the run raises ValueError listing the valid names. Default 'predictions'.",
"homogeneity": "(bool) - Compute grey-level co-occurrence-matrix homogeneity for every object in every channel, adding one homogeneity_distance_<d> column per entry in homogeneity_distances. Homogeneity is high for smooth, evenly filled objects and low for punctate or grainy ones, so keep it on for texture phenotypes; disabling it noticeably speeds up measurement. Default True.",
"homogeneity_distances": "(list) - Pixel offsets used to build each object's grey-level co-occurrence matrix; every entry adds one homogeneity_distance_<d> feature per channel. Small offsets capture fine-grained texture, large ones capture coarse structure, and offsets larger than the object itself carry no signal. More entries means more features and slower measurement. Default [8, 16, 32].",
"image_nr": "(int) - How many example object crops to draw on the embedding plot: that many per cluster when plot_by_cluster is on (smaller clusters show all they have), otherwise that many sampled at random overall. It also sets how many images each cluster contributes to the cluster-grid figure. Raise it for a fuller montage, lower it when thumbnails hide the points. Default 16.",
"image_size": "(int) - Side length in pixels of the center crop taken from each object PNG before model input. Images are cropped rather than rescaled, so larger values add zero padding and smaller values discard peripheral object pixels. This value also defines the backbone input resolution and must match the crop size used to generate the dataset. Default 224.",
"img_zoom": "(float) - Scale applied to each object thumbnail pasted onto the embedding: 1.0 draws the crop at native pixel size, 0.5 at half. Raise it when crops are too small to judge morphology, lower it when thumbnails overlap and bury the point cloud. Practical range about 0.1-2.0. Default 0.5.",
"uninfected": "(bool) - Decides which cells survive the consistency filter in measure_crop. True keeps any cell that has both a nucleus and a cytoplasm; False also demands at least one pathogen, dropping uninfected cells from every table. Either way, nucleus/pathogen/cytoplasm labels outside the surviving cells are zeroed. Only applied when cell, nucleus and pathogen masks all exist; forced True otherwise. Default True.",
"init_weights": "(bool) - Start the backbone from ImageNet-pretrained weights instead of random initialisation; the spaCR classifier head bolted on top is randomly initialised either way. Leave it on - transfer learning converges in far fewer epochs on the small annotated sets typical here. Turn it off only to train from scratch on a very large dataset, or to measure how much pretraining contributes. Default True.",
"intermedeate_save": "(bool, sequence of float, or None) - Control archival model snapshots on improving epochs. True or None uses validation-accuracy thresholds 0.99, 0.98, 0.95 and 0.94; False disables archival snapshots; and a sequence supplies custom thresholds. Best-model and last-model checkpoints remain enabled independently, so False does not remove those recovery artifacts. Default True.",
"invert": "(bool) - Invert intensities as each image is loaded, pixel -> dtype_max - pixel (255 - x for uint8). Switch it on for brightfield or phase-contrast data where objects are darker than the background, since Cellpose expects bright objects on a dark field; leave it off for fluorescence. Default False.",
"learning_rate": "(float) - Initial optimizer step size. Values that are too high may prevent convergence; values that are too low may slow convergence or converge to a suboptimal solution. A value near 1e-3 is commonly used for training from random initialization, while 1e-4 to 1e-5 is appropriate for fine-tuning ImageNet weights (init_weights=True). The selected schedule modifies this initial value during training. Default 0.001.",
"location_column": "(str) - Metadata column searched for positive_control_id and negative_control_id values when labelling rows for machine-learning training, normally 'columnID' or 'rowID'. Set 'rowID' when controls are arranged along plate rows instead of columns. annotation_column overrides this setting when specified. Default 'columnID'.",
"log_data": "(bool) - Apply log(x + 1e-6) to every numeric feature, after the correlation filter and before standard scaling. Compresses heavy-tailed measurements such as intensity sums and areas so a handful of bright or huge objects stop dominating the embedding. Negative feature values become NaN and are then filled with the column mean. Default False.",
"lower_percentile": "(float) - Percentile of the non-zero pixels in each channel used as the low anchor when rescaling that channel to 0-1; the high anchor is chosen automatically between the 98th and 99.5th percentile. Raise it to crush more dim background to black, lower it to preserve faint signal. Valid 0-100, default 2.",
"manders_thresholds": "(list) - Percentiles (0-100) used by the activation-map correlation report in spacr.deep_spacr. It no longer affects a measure run: the percentile-pair columns it drove there were removed on 2026-09-02, and measure now writes the three standards-compliant Manders coefficients, which estimate each channel's background inside each object and take no percentile. Default [15, 50, 75].",
"mask": "(bool) - Whether to generate masks for the segmented objects. If True, masks will be generated for the nucleus, cell, and pathogen.",
"measurement": "(str) - Measurement column(s) from measurements.db used to prefilter which object crops the annotator loads, applied together with threshold and threshold_direction. Accepts a single column, a comma-separated list (each paired with the same-index threshold), or a JSON list-of-lists where an inner pair is filtered as a ratio (first divided by second). Empty (default) loads every crop unfiltered.",
"merge_edge_pathogen_cells": "(bool) - During measurement, reconcile pathogens straddling two host-cell masks: if 90 percent or more of the pathogen lies in one cell, its pixels in the neighbours are erased; otherwise the overlapping cell labels are fused into a single cell. Switch off to keep the raw cell segmentation when parasites legitimately touch two cells. Default True.",
"metric": "(str) - Distance metric used both by the reducer (UMAP or t-SNE) and by DBSCAN clustering, e.g. 'euclidean', 'manhattan', 'cosine' or 'correlation'. Correlation-type metrics compare feature profiles regardless of magnitude and often separate phenotypes better than euclidean on scaled data. Default 'euclidean'.",
"min_cells_per_well": "(int) - Wells with fewer than this many cells are dropped. In a regression it is scored objects and the well is left out of the fit; in the machine-learning screen it is measured cells and the well is left out of the plate heatmap, whose pivot is then filled with 0, so an excluded well renders at the bottom of the colour scale rather than blank. Raising it removes noisy, sparsely imaged wells at the cost of power. Set 0 to switch it off. Default 100 for a regression, 25 for the screen.",
"min_dist": "(float) - UMAP's minimum spacing between points in the 2-D embedding, range 0.0-1.0. Low values (0.0-0.1) let clusters pack tightly and look crisply separated; higher values spread points out and preserve more of the global layout at the cost of visible cluster structure. Ignored when reduction_method is 'tsne'. Default 0.1.",
"tsne_perplexity": "(float) - t-SNE neighborhood scale. It must be smaller than the number of rows; values around 5-50 are typical. Low values emphasize very local structure and can fragment populations; high values smooth them together. Used only by t-SNE. Default 30.",
"tsne_learning_rate": "(float) - t-SNE optimization step size. Too small crowds points into a dense ball; too large can scatter them. Used only by t-SNE. Default 200.",
"tsne_early_exaggeration": "(float) - t-SNE's initial attraction multiplier, controlling how much space forms between natural groups early in optimization. Used only by t-SNE. Default 12.",
"tsne_max_iter": "(int) - Maximum t-SNE optimization iterations. Increase it when optimization has not stabilized; every increase costs runtime. Used only by t-SNE. Default 1000.",
"pca_whiten": "(bool) - Rescale PCA components to unit variance after projection. This can help distance-based clustering but discards relative component magnitude. Used only by PCA. Default False.",
"pca_svd_solver": "(str) - PCA decomposition algorithm: auto chooses from the data shape, full is exact, randomized is faster on large matrices, and covariance_eigh suits many rows with relatively few features. Used only by PCA. Default 'auto'.",
"isomap_n_neighbors": "(int) - Number of neighbors in Isomap's geodesic graph. Too few can disconnect the graph; too many make the result approach a global linear projection. Used only by Isomap. Default 15.",
"isomap_path_method": "(str) - Isomap shortest-path solver: auto chooses, FW uses Floyd-Warshall, and D uses Dijkstra. Used only by Isomap. Default 'auto'.",
"spectral_affinity": "(str) - Graph construction for Spectral Embedding: nearest_neighbors builds a sparse local graph; rbf builds a dense radial-basis affinity. Used only by Spectral Embedding. Default 'nearest_neighbors'.",
"spectral_n_neighbors": "(int) - Neighbor count for Spectral Embedding when affinity is nearest_neighbors. Ignored for rbf affinity. Default 15.",
"gpu": "(bool) - Request RAPIDS acceleration for the main dimensionality reduction and Image UMAP hyperparameter search. Controlled by the GPU toggle beside Hyperparameter search; supported for UMAP, t-SNE and PCA, with the actual backend recorded. Default False.",
"min_max": "(str) - Color limits for the plate heatmap: 'allq' scales to the 2nd-98th percentile of well values so a handful of extreme wells cannot flatten the rest, 'all' scales to the true min and max. A two-element list is also accepted, where floats are read as quantiles and integers as absolute vmin/vmax. Default 'allq'.",
"min_samples": "(int) - Meaning depends on 'clustering': for DBSCAN it is how many points must fall within eps for a point to count as a core point, so raising it yields fewer, denser clusters and more noise; for KMeans this same value is reused as n_clusters, the exact number of clusters produced. Lower it (or raise eps) when no clusters are found. Default 100.",
"mix": "(str) - Plate column ID whose wells hold a mixed positive/negative population; rows with this columnID are labelled cond='mix' for the image UMAP, so they can be coloured separately or dropped via exclude_conditions. Any column matching none of pos, neg or mix is labelled 'screen'. Default 'c3'.",
"model_name": "(str) - Cellpose model used for segmentation. Cellpose 4 provides one stock model, 'cpsam'. Pre-SAM names ('cyto', 'cyto2', 'cyto3', 'nuclei') remain accepted for compatibility with older settings, but they are mapped to 'cpsam' and reported. Of the three parameters that previously distinguished models, only diameter remains operational in Cellpose 4 (eval rescales the image by 30/diameter); model_type and diam_mean are logged as 'not used in v4.0.1+' and omitted. Use 'cpsam' unless loading a custom CPSAM checkpoint. Default 'cpsam'.",
'mask_src': '(str) - Optional separate folder of integer object-label masks (background 0). Leave blank to use masks inside the image folder. Match image basenames, optionally with a _masks suffix. Missing or ambiguous pairs stop training. Default blank.',
'test_src': '(str) - Optional validation image folder, separate from training images. Leave blank to train without validation losses. Split by well or experiment to avoid leakage between related fields. Default blank.',
'test_mask_src': '(str) - Optional validation label-mask folder. Leave blank to use the masks subfolder inside the validation image folder. Default blank.',
'save_path': '(str) - Checkpoint output folder. Cellpose writes weights into its models subfolder. Leave blank for <image source>/models/cellpose_model. Use trained model reads this location too. Default blank.',
'channel_axis': '(int or None) - Image channel axis: 0 for channel-first, -1 for channel-last, or blank to infer it from the mask dimensions. Ambiguous images require an explicit axis. Training supports 2-D fields with optional channels, not Z stacks. Default None (automatic).',
'min_train_masks': '(int) - Minimum labeled objects required per training image. Cellpose excludes fields below this count. Default 5. Lower it for deliberately sparse training fields.',
'max_train_images': '(int or None) - Optional limit on paired training images loaded into RAM, in filename order. Blank or a nonpositive value uses every pair. This does not change the minibatch size. Default None (automatic).',
'nimg_per_epoch': '(int or None) - Optional number of images sampled per training epoch. Blank uses every training image. This changes sampling, not the number of files loaded into RAM. Default None (automatic).',
'nimg_test_per_epoch': '(int or None) - Optional number of validation images sampled per evaluation epoch. Blank uses all validation images. Requires a validation image source. Default None (automatic).',
'scale_range': "(float) - Range of Cellpose's random training scale augmentation, from 0 to 2. Default 0.5. Cellpose also applies its native rotation, flip and crop augmentation; no eight-fold duplicate dataset is created.",
'save_every': '(int) - Checkpoint interval in epochs. Default 100. Cellpose always saves the final model even when the run is shorter than this interval.',
'save_each': '(bool) - Keep separate epoch checkpoints instead of replacing the periodic checkpoint. Default False. Enable to compare intermediate models; it consumes additional disk space.',
"base_model": "(str) - Train Cellpose: the weights training starts from. 'cpsam' is stock Cellpose-SAM; a model-zoo key (for example 'toxoplasma_plaque_v2') or a path to a checkpoint continues from that model, which is how a second fine-tuning stage builds on the first. The model that was started from is recorded with the run's settings. Default 'cpsam'.",
"model_type": "(str) - Backbone architecture for the single-object image classifier: any TorchVision classification model name (resnet50, maxvit_t, densenet121, ...). An unrecognized name does not fail during initial validation: choose_model reports 'Invalid model_type' and returns None, after which training fails. The special name 'custom' passes validation and then raises NotImplementedError. Larger backbones require more memory and generally need more labeled crops than smaller backbones. Default 'maxvit_t'.",
"model_type_ml": "(str) - Classifier fitted by ml_analysis to separate positive- from negative-control wells and rank per-object features by permutation importance. Options are xgboost (default), lightgbm, catboost, random_forest, extra_trees, gradient_boosting, logistic_regression, svm and mlp; lightgbm and catboost require their optional packages. reg_alpha, reg_lambda and learning_rate affect only boosted models; logistic_regression provides a linear reference model.",
"negative_control_id": "(str) - Identifier of the negative-control class. In ML screening it is the value in location_column (e.g. 'c1') whose objects are labelled class 0 for training; in gRNA regression it is a gene/gRNA ID substring (e.g. '233460') matched against coefficient names to tag them 'nc' in the results and volcano plot. Defaults 'c1' and '233460' respectively.",
"n_estimators": "(int) - Number of trees or boosting rounds in the tabular ML classifier - n_estimators for RandomForest/ExtraTrees/XGBoost/LightGBM, iterations for CatBoost, max_iter for HistGradientBoosting. More rounds keep improving fit up to a plateau while training time grows linearly; boosted models can overfit past it. Default 1000.",
"n_epochs": "(int) - Number of training passes train_seg makes over the annotated image/mask batch. It also sets the checkpoint interval (a model is saved every n_epochs/10) and is written into the saved model filename. Raise it for a better fit on large annotation sets; lower it when a small set starts overfitting. Default 10000.",
"n_neighbors": "(int or float) - Size of the local neighbourhood UMAP balances against global structure, and the perplexity when reduction_method is 'tsne'. Small values (5-50) sharpen fine local structure; large values give a smoother, more global embedding. A float is read as a fraction of the number of objects, and anything below 2 is clamped to 2. Default 1000.",
"n_repeats": "(int) - Number of random shuffles per feature when computing permutation importance for the ML classifier. More repeats reduce uncertainty in the importance ranking but require an additional prediction pass per feature and repeat. Default 10; use 3-5 for a faster preliminary assessment of wide feature tables.",
"pathogen_signal_to_noise": "(int) - Expected foreground-to-background ratio of the pathogen channel. Multiplied by pathogen_background, it defines the minimum intensity for the normalisation ceiling. spaCR evaluates percentiles 98 through 99.5 and uses the first that reaches the threshold, with the 99.5th percentile as the fallback. Increase it for a higher ceiling with less clipping; decrease it for greater contrast. Default 20.",
"nucleus_signal_to_noise": "(float) - Multiplied by nucleus_background to define the intensity a bright pixel must reach before normalisation stops increasing the upper clip point. spaCR evaluates the 98th through 99.5th percentiles of the non-zero nucleus channel and uses the first that reaches the threshold, with the 99.5th percentile as the fallback. A higher value raises the clip point, reduces contrast stretching and protects bright nuclei from saturation; a lower value increases contrast for dim nuclei but saturates bright nuclei sooner. Default 10.",
"pathogen_size_range": "(list) - Two-element [min, max] area filter in pixels squared applied to the pathogen table in analyze_recruitment, well after segmentation: rows with pathogen_area outside the open interval are dropped. Bounds must be ints - floats are silently ignored. None widens it to effectively unlimited. Default [0, 100000]. Use it to discard debris and merged clumps.",
"pathogen_types": "(list) - Names given to each pathogen condition on the plate, e.g. ['wt','ku80']. Element i is written into the pathogen column for every well listed in pathogen_plate_metadata[i] and folded into the combined condition label used for grouping and plotting. Must match pathogen_plate_metadata in length and order; None skips pathogen annotation. Default ['pathogen_1', 'pathogen_2'] for the dataset builders, ['pc'] for the control-based paths, None where types are not used.",
"percentiles": "(list) - Two percentiles [low, high] used to rescale each channel of each image to 0-1 before segmentation, e.g. [2, 98]. Narrowing the window boosts contrast on dim objects but clips bright ones. Set None to derive them automatically: low fixed at 2, high the first of 98/99/99.9/99.99/99.999 exceeding background * Signal_to_noise. In Cellpose Masks (Apply), None instead lets Cellpose normalise each image itself, as the live preview does. Default None in the Cellpose steps.",
"pin_memory": "(bool) - Decode and hold the entire train/test image set in RAM up front (loaded in parallel across all cores) and hand batches to the GPU from page-locked memory. Enable when the dataset fits comfortably in RAM and disk I/O is the bottleneck; disable for large datasets or it will exhaust memory before the first epoch even starts. Default False.",
"plate": "(str) - Legacy setting that is not read by the regression path. Use plateID instead; perform_regression passes plateID to process_scores and process_reads, which apply it to count and score rows lacking a plate identifier. Default None.",
"plot": "(bool) - Render and save quality-control figures during the pipeline, including channel montages, Cellpose mask overlays, filtration comparisons, and crop grids. Figure generation increases runtime and memory use, particularly for complete plates. test_mode enables this setting automatically. Default False. Merged Classifier and Recruitment both start with plotting enabled so their diagnostic figures are produced on the first run.",
"plot_by_cluster": "(bool) - Chooses which thumbnails get overlaid on the embedding: when True, up to image_nr crops are sampled from each cluster (DBSCAN noise excluded) so every cluster is represented; when False, image_nr crops are sampled at random across the whole map. Keep True to compare cluster morphologies, False for an unbiased sample. Default True.",
"plot_cluster_grids": "(bool) - Render a second figure with one colour-bordered panel per cluster, each containing up to image_nr example crops, and save it as <METHOD>_grid.pdf when save_figure is enabled. The cluster grid is emitted after the embedding and therefore becomes the final displayed figure. Ignored unless plot_images is True. Default False.",
"plot_control": "(bool) - Before the recruitment plots, draw a control panel of per-compartment mean intensities (cell, nucleus, pathogen, cytoplasm) for every channel, split by condition. Use it to confirm channel assignment and that positive/negative control wells separate as expected before trusting the recruitment numbers. Turn it off to shorten the run. Default True.",
"plot_images": "(bool) - Paste the actual object crops onto the embedding scatter instead of showing bare points. Turn it off for a fast, plain scatter on large datasets - doing so also forces black_background to False and skips the cluster grid figure entirely. Default True.",
"plot_nr": "(int) - Number of merged image stacks from the start of the folder displayed with cell, nucleus, and pathogen outlines before recruitment analysis. The implementation checks index <= plot_nr, so plot_nr + 1 images are displayed and 0 displays one image. Increase the value to inspect segmentation across more fields. Default 3.",
"plot_outlines": "(bool) - Draw a boundary around each cluster in the embedding - a smoothed hull when smooth_lines is True, otherwise the raw convex hull edges. Helps show cluster extent and overlap but clutters dense maps; clusters with fewer than three points are skipped. Forced off when color_by is set. Default True.",
"png_dims": "(list of int) - Deprecated in favor of png_channel_mapping and retained for compatibility with older settings files. Under the legacy mapping, entry 0 becomes blue, entry 1 green and entry 2 red, matching the wavelength order 0=405, 1=488 and 2=555. Ignored when png_channel_mapping is set. Default [].",
"png_channel_mapping": "(dict) - Which source channel goes in each colour of the saved PNG, e.g. {'r': 2, 'g': 1, 'b': 0}: channel 2 is red, 1 is green, 0 is blue. Says outright what png_dims only implied. Channels not named are absent from the crops (measurements are unaffected); a colour left blank is an empty plane. Naming the same channel for all three writes a greyscale PNG. Default {'r': 2, 'g': 1, 'b': 0}, which for a standard 405/488/555 stack puts the nuclear stain in blue.",
"png_size": "(list of int) - Output crop size as [width, height] in pixels, centred on the object centroid; larger keeps more surroundings, smaller clips large objects. Should match the classifier input size (default [224,224]). With several crop_mode entries pass a list of lists, one size per mode, or a single size is reused for all.",
"positive_control_id": "(str) - Identifier of the positive-control class. In ML screening it is the value in location_column (e.g. 'c2') whose objects are labelled class 1 for training; in gRNA regression it is a gene/gRNA ID substring (e.g. '239740') matched against coefficient names to tag them 'pc' in the results and volcano plot. Defaults 'c2' and '239740' respectively.",
"preprocess": "(bool) - Run image preparation before segmentation: group raw files into per-field channel stacks, optionally subtract background, and percentile-normalize each channel into floating-point arrays. Keep True for unprocessed input; set False only when the normalized arrays already exist, because segmentation requires those arrays. Default True.",
"confluency": "(bool) - Measure confluency, the fraction of each field covered by cells, and write it to measurements.db: one row per field in the confluency table and one per well in confluency_well, with a monolayer_ok flag the plaque and infection assays can filter on or divide by. Works for any channel (brightfield, phase or a fluorescent stain) or straight from the cell masks, as confluency_source decides. With plot on, each field also gets an overlay of the covered area. Default False.",
"confluency_source": "(str) - How confluency is decided. auto uses the cell masks when the run has cell masks and texture otherwise. masks is the union of every segmented cell, before Measure's size filters. texture reads the local variation of confluency_channel with an automatic threshold. phase classifies every pixel of confluency_channel with a small model, for phase contrast and brightfield; weights trained on LIVECell, CC BY-NC 4.0, non-commercial use. intensity thresholds confluency_channel automatically, for fluorescent cytoplasm or membrane stains. Default auto.",
"confluency_channel": "(int or None) - The merged-array channel that the texture, intensity and phase confluency sources read, counted as in channels. Blank uses the first entry of channels. Pick the brightfield or phase plane for texture or phase, or the cytoplasm or membrane stain for intensity. Ignored when confluency_source resolves to masks. Default None.",
"confluency_window": "(int) - Side of the square window, in pixels, over which the texture confluency source measures local variation. Roughly the width of the thinnest cell process that should count as covered: smaller follows edges more closely but leaves smooth cell interiors as holes, larger bridges narrow gaps. For phase, the field is first resized by 15/window, so raise it in proportion when cells are more pixels across than in the classifier's training images. Ignored by the masks and intensity sources. Default 15.",
"bleach_correction": "(str) - Photobleaching correction for a timelapse run, applied after measuring and per field and channel. ratio rescales each timepoint so the median object mean intensity equals the first timepoint's; exponential does the same with a fitted a*exp(-b*t)+c decay; histogram maps each timepoint's intensities onto the first timepoint's distribution. Writes <object>_bleach_corrected and the fits to measurements.db and plots the decay; the measured tables stay unchanged. Ignored unless timelapse. Default none.",
"measure_gpu": "(bool) - Compute the per-object intensity statistics, GLCM homogeneity and Zernike moments on a CUDA GPU through PyTorch, all objects of a field at once instead of one at a time. With cuCIM installed (the 'gpu' extra), the per-object morphology table is computed on the GPU too; without it, morphology stays on the CPU. Values match the CPU run within float tolerance. Covers 2-D masks without voxel spacing; anything else, a missing PyTorch or no visible CUDA device measures on the CPU as usual. Default False.",
"measurement_backend": "(str) - Where a finished run's measurements are also stored. sqlite keeps only measurements.db. duckdb copies every table into a DuckDB file and parquet into a folder of Parquet files, both for very large screens; postgres copies them into a PostgreSQL database that several users can write at once. measurements.db stays the working copy every later step reads. Needs pip install spacr[databases]. Default sqlite.",
"measurement_backend_target": "(str) - The DuckDB file, Parquet folder or PostgreSQL connection string the measurements are copied to. Blank puts measurements.duckdb or measurements.parquetdb beside measurements.db, and reaches PostgreSQL through the PGHOST, PGDATABASE, PGUSER and PGPASSWORD environment variables. Keep passwords in ~/.pgpass, not here. Ignored for sqlite. Default blank.",
"wound_closure": "(bool) - Measure a scratch or wound-healing assay: find the open wound in every frame of every field, then write its area, mean and minimum width, the closure rate and the half-closure time per field, per well and per condition to measurements.db and results/wound_closure, with closure curves and a plate map. Frames are grouped by plate, well and field and ordered by timepoint; the first frame decides where the scratch is. Default False.",
"wound_source": "(str) - How the open wound is told apart from the monolayer. texture reads the local variation of wound_channel, for brightfield and phase. intensity thresholds wound_channel, for a fluorescent cytoplasm or membrane stain. masks takes every pixel outside the segmented cells as open. The cut is decided on each field's first frame and kept for its later frames. Default texture.",
"wound_channel": "(int or None) - The merged-array channel the texture and intensity wound sources read, counted as in channels. Blank uses the first entry of channels. Pick the brightfield or phase plane for texture, the stain for intensity. Ignored by the masks source. Default None.",
"wound_window": "(int) - Side of the square window, in pixels, over which the texture wound source measures local variation. About the diameter of one cell at the imaging resolution: smaller follows the wound edge more closely but can open holes in smooth parts of the monolayer, larger bridges narrow gaps. Also sets the smallest gap kept in later frames. Default 15.",
"wound_threshold": "(float or None) - Sets the cut between open wound and monolayer by hand, on the scale of the wound_level column of the wound table: for texture the local variance over the field's median, for intensity a share of the frame's 95th percentile. Use it when the automatic cut misreads a series: run once, read wound_level, then set a value. Higher counts more of the field as open. Used on every frame, with no recalibration of later frames. Blank or 0 keeps the automatic cut. Ignored by the masks source. Default None.",
"wound_hours_per_frame": "(float or None) - Hours between consecutive timepoints, so closure rates are per hour and half-closure times are in hours. Blank counts time in frames. With voxel_size_xy_um set, widths and front speeds are also reported in micrometres. Default None.",
"wound_conditions": "(dict) - Conditions to pool wells into for the closure curves and half-closure times, as {name: wells}, the wells as rows (r2), columns (c3) or single wells (B03), for example {'control': 'c1, c2', 'drug': 'c3, c4'}. A well in no condition is reported under its own name. A well may belong to one condition only. Default {}.",
"confluency_qc_threshold": "(float or None) - Lowest covered fraction, from 0 to 1, at which a field or well passes monolayer QC. Fields and wells below it get monolayer_ok 0 in measurements.db, so plaque and infection results from a thin or torn monolayer can be dropped or divided by the covered fraction. Blank passes every well. Default 0.8.",
"profiling": "(bool) - After Measure finishes, build image-based profiles from its tables: aggregate each object table to one median profile per well, add the plate map in profiling_metadata, normalise each plate against its negative-control wells, remove uninformative and redundant features, build one consensus profile per treatment and score replicate reproducibility as mean average precision (mAP) and percent replicating. Results go to measurements/profiles as CSV, Parquet and GCT with plots. Default False.",
"profiling_metadata": "(str) - Plate map for profiling: a CSV, TSV, Excel or Parquet table with one row per well position, located by rowID and columnID or by a well column such as A01, plus annotation columns such as treatment, dose, gene or a phenotype label. With a plateID column the map is matched plate by plate; without one it applies to every plate. Blank profiles the wells by position only. Default blank.",
"profiling_treatment_column": "(str or list) - The annotation column, or columns, naming what each well received. Wells that share them are replicates and are collapsed into one consensus profile. Use a plate-map column such as treatment, treatment and dose together, or columnID when each plate column holds one condition. Default columnID.",
"profiling_negative_control": "(str or list) - Value or values of the first profiling_treatment_column that mark negative-control wells, for example DMSO or c1. Each plate is normalised against its own controls, and every treatment is scored for phenotypic activity, how well its replicates find each other among the controls. Blank normalises against all wells of a plate and skips the activity score. Default blank.",
"profiling_normalization": "(str) - How each feature is put on a common scale, plate by plate. mad_robustize subtracts the reference median and divides by 1.4826 times the reference MAD, which tolerates outlier wells; standardize uses the mean and standard deviation; robustize uses the median and interquartile range; none keeps the aggregated values. The reference is the negative control, or every well when none is named. Default mad_robustize.",
"profiling_feature_selection": "(list) - Feature-selection steps, run in order after normalisation: variance_threshold drops near-zero variance, frequency_threshold near-constant values, correlation_threshold the more redundant of each pair correlated above profiling_correlation_threshold, drop_na_columns features missing in over 5 % of wells, and drop_outliers any feature with an absolute value above 500. An empty list keeps every feature. Default all five.",
"profiling_correlation_threshold": "(float) - Pearson correlation above which two features count as redundant in the correlation_threshold selection step. Of each such pair the feature more correlated with all others is removed. Lower values keep fewer, less redundant features. Default 0.9.",
"profiling_phenotype_column": "(str) - Optional plate-map column holding a phenotype label that different treatments share, such as a mechanism of action, pathway or target gene. When set, consensus profiles are also scored for phenotypic consistency: how well treatments with the same label retrieve each other, as mAP per label. Default blank.",
"profiling_databases": "(list) - Further measurements.db files to profile together with this run's, one per plate, so replicates on different plates are compared while each plate is still normalised on its own. Every plate needs a distinct plateID. Default [].",
"cell_cycle": "(bool) - Call the cell-cycle phase of every nucleus after measuring, from the DNA stain: G1, S, G2 or M, plus subG1 and >4N outside the peaks. Writes one row per nucleus to measurements.db:cell_cycle with the call in cell_cycle_phase, the phase fractions per well, overall and in infected and uninfected cells, to cell_cycle_well, and with plot on each plate's fitted DNA histogram. Needs measured nuclei. Default False.",
"cell_cycle_method": "(str) - How phases are called. measurements gates each plate's DNA-content histogram, fitted as G1 and G2 peaks with S between, and calls condensed 4N nuclei M. xgboost trains a boosted classifier on nucleus features, torch trains an image classifier on nucleus crops with Classify's training; both learn from cell_cycle_labels. all runs the three and keeps their majority. Default measurements.",
"cell_cycle_channel": "(int or None) - The merged-array channel holding the DNA stain (DAPI or Hoechst) the phase is read from, counted as in channels; it must be one of the measured channels. Blank uses nucleus_channel, then the first entry of channels. Default None.",
"cell_cycle_gates": "(list or None) - Fixed gates [G1/S, S/G2] in DNA content units, where the G1 peak is 2 and the G2 peak 4, for example [2.5, 3.5]. They replace the crossings fitted per plate; the peaks are still fitted to place the units. Read the fitted gates off the saved histogram before editing them. Blank uses the fitted gates. Default None.",
"cell_cycle_mitotic_ratio": "(float or None) - A nucleus past the G1/S gate is called M when its background-subtracted mean DNA intensity is at least this many times the median of the plate's G2 nuclei: condensed mitotic chromatin is brighter. Lower catches more prophase and more bright G2 nuclei. Blank never calls M from intensity and leaves mitotic nuclei in G2. Default 1.8.",
"cell_cycle_fucci_channels": "(list or None) - Two merged-array channels of a FUCCI reporter, the G1 reporter (Cdt1) then the S/G2/M reporter (Geminin), both measured. Each is split into positive and negative per plate, and every nucleus gets a fucci_state: early G1, G1, G1/S or S/G2/M. The xgboost method also uses their intensities. Blank skips FUCCI. Default None.",
"cell_cycle_labels": "(str) - A png_list column holding Annotate labels the xgboost and torch methods learn from: 1 to 4 for G1, S, G2 and M, or the phase names. Labels are matched to nuclei through their cell. Blank trains on the confident gate calls instead, which teaches the learned methods what the gates already say; annotate prophase and anaphase nuclei to teach them more. Default blank.",
"cell_cycle_model": "(str) - A torch model this step trained earlier, with its cell_cycle_phases.json beside it, applied to the nucleus crops instead of training a new one. Use it to call a second plate with the model trained on the first. Ignored unless cell_cycle_method is torch or all. Default blank.",
"cell_cycle_epochs": "(int) - Training epochs of the torch phase classifier. It is a ResNet-18 trained from scratch on crops as small as 32 pixels, so each epoch is quick on a GPU and slow on a busy CPU. Ignored when cell_cycle_model names a trained model. Default 20.",
"intensity_calibration": "(bool) - Calibrate intensities across imaging sessions before measuring: each plate is one session, the beads or reference wells imaged on every plate are measured, and every plate's intensity channels are scaled so its reference wells match the first plate's. This corrects exposure, lamp and detector drift between days at the image level, unlike batch correction of tables. The gains are recorded in measurements.db:intensity_rescale. Default False.",
"intensity_calibration_wells": "(list or None) - The wells holding the calibration sample, imaged on every plate with the same sample: fluorescent beads or a reference stain, such as ['A01'] or ['A01', 'P24']. Every plate must have at least one field in them, or the run stops. They are measured and calibrated like any other well. Default None.",
"intensity_calibration_statistic": "(str) - How each reference field's intensity is summarised, after subtracting intensity_calibration_offset. foreground: the median of the pixels above an Otsu threshold, for sparse beads on a dark background. median: the median of all pixels, for a uniformly stained reference well. Each plate uses the median over its reference fields. Default foreground.",
"intensity_calibration_offset": "(float) - The camera's dark offset, in the intensity units Measure works in, removed before the reference statistic and kept when scaling: a pixel becomes offset + (value - offset) x gain. Read it from a dark frame; 100 is common on sCMOS cameras. Leave 0 when images are already offset-corrected. Default 0.",
"plate_barcode_source": "(str) - Sample records to fill the plate map from, by plate barcode: a CSV, TSV, Excel or Parquet table with a barcode column, a well column (well such as A01, or rowID and columnID) and metadata such as strain, compound, concentration, passage and operator; or the http(s) address of a LIMS service that answers JSON well records for ?barcode=, or for {barcode} in the address. The filled map and a list of mismatches go to measurements. Blank links nothing. Default blank.",
"plate_barcodes": "(dict or None) - The barcode each plate was imported with, such as {'plate1': 'BC000123'}. A plate not named here takes the barcode in a barcode.txt file in its plate folder, or else its own plate name. Two plates given one barcode are reported as a mismatch. Default None.",
"plate_barcode_column": "(str) - The column of the sample records that holds the plate barcode. Blank looks for barcode, plate_barcode or plate barcode. The other columns, apart from the well position, are copied into the plate map as they are. Default barcode.",
"plate_barcode_token_env": "(str) - The name of the environment variable holding the LIMS access token, sent as a bearer token with each request. The token itself is never written to settings or results; set the variable before starting spaCR. Ignored for a table. Default SPACR_LIMS_TOKEN.",
"time_to_event": "(bool) - After measuring a timelapse, follow every tracked object to an event (death, lysis, egress, division, first detection) or to the end of its track, and compare conditions: Kaplan-Meier curves with 95% bands, median time to event per condition and well, log-rank tests and a Cox model. Writes measurements.db:time_to_event and four summary tables, and the curves and hazard ratios under results/time_to_event. Needs tracked objects measured with timelapse on. Default False.",
"time_to_event_object": "(str) - The measured object table whose tracks are followed: cell, nucleus, pathogen or cytoplasm. Each object label in a field is one track, since the timelapse module relabels tracked objects with their track ID. Follow host cells for host death or lysis, pathogens for egress or division. Default cell.",
"time_to_event_mode": "(str) - What counts as the event. track_end: the object disappears before the movie ends (lysis, egress, detachment, and tracking loss too). annotated: time_to_event_column turns non-zero, or equals the threshold. above or below: the column reaches time_to_event_threshold, such as a death dye. fold_change: the column reaches threshold times its first value, such as a doubled parasite count. Tracks without the event are censored at their last frame. Default track_end.",
"time_to_event_column": "(str) - The measurement the event is read from, a column of the object table such as cell_channel_2_mean_intensity, or an Annotate column of png_list holding labels for each object in each frame. Ignored by track_end. Default blank.",
"time_to_event_threshold": "(float or None) - The cut the event is read at. For above and below it is in the column's units; for fold_change it is a ratio to the track's first value, 2 for a doubling; for annotated it is the label that marks the event, blank for any non-zero label. Read it off a histogram of the column before trusting the curves. Default None.",
"time_to_event_persist": "(int) - Consecutive observed frames the event must hold before it counts; it is dated to the first of them. Raise it to 2 or 3 when a noisy measurement crosses the threshold for one frame and back. Objects already showing the event in their first frame are left out. Default 1.",
"time_to_event_origin": "(str) - Where each object's clock starts. track: at the object's own first frame, so objects that appear later (daughters, cells moving in) count from when they are first seen. movie: at the movie's first frame, keeping only the objects present then, the cohort to use when time since infection or treatment matters. Default track.",
"time_to_event_min_frames": "(int) - Tracks with fewer observed frames are left out: a track seen once or twice is more often a segmentation or tracking fragment than an object, and with track_end every short fragment would be an event. Default 3.",
"time_to_event_hours_per_frame": "(float or None) - Hours between consecutive frames, used to report times in hours; 0.25 for a frame every 15 minutes. It rescales the times and medians but not the tests or hazard ratios of conditions. Blank reports times in frames. Default None.",
"time_to_event_group": "(str) - How objects are grouped into the conditions compared, when time_to_event_conditions is blank: well, plate, row, column, field, or a column of the object table read at each track's first frame. Default well.",
"time_to_event_conditions": "(list or None) - Conditions named by their wells, as name=wells entries in the plate-map notation, such as ['mock=c1,c2', 'drug=c3,c4']. Objects in wells no entry names are left out. It replaces time_to_event_group. Blank uses time_to_event_group. Default None.",
"time_to_event_reference": "(str) - The condition the others are compared with: every log-rank pair and every hazard ratio is against it. It must be one of the conditions. Blank uses the first entry of time_to_event_conditions, or the first condition in sorted order. Default blank.",
"time_to_event_covariates": "(list or None) - Columns of the object table adjusted for in the Cox model, each read at the track's first frame so it is measured before the event, such as ['cell_area']. Their hazard ratios are per unit of the column. Tracks missing a value are left out of the model only. Blank fits the conditions alone. Default None.",
"viability": "(bool) - Call every cell live or dead after measuring, from a dead stain (propidium iodide, SYTOX, DAPI on unfixed cells), a live stain (calcein), both, or without either from nuclear morphology (pyknotic nuclei). Writes one row per nucleus to measurements.db:viability, per-well viability, live-cell and cytotoxicity index to viability_well, each plate's thresholds and control Z' to viability_qc, and with plot the threshold, plate, control and dose-response figures. Default False.",
"viability_dead_channel": "(int or None) - The merged-array channel of the dead stain (propidium iodide, SYTOX, or DAPI added to unfixed cells), counted as in channels; it must be one of the measured channels. Each nucleus's background-subtracted mean intensity is split per plate into two populations, and above the cut is dead. Blank reads no dead stain; with neither stain channel set, dead cells are called from nuclear morphology instead. Default None.",
"viability_live_channel": "(int or None) - The merged-array channel of the live stain (calcein-AM), counted as in channels; it must be one of the measured channels. It is read on the nucleus, which calcein fills, split per plate, and above the cut is live. With a dead stain as well, a cell positive for neither is counted unstained, not live. Blank reads no live stain. Default None.",
"viability_thresholds": "(list or None) - Manual cuts [dead, live] in background-subtracted mean intensity, for example [150, None], each replacing the automatic per-plate cut of its stain; None keeps that one automatic. Without stain channels the dead entry is the condensation ratio (intensity per area against the plate's typical nucleus, about 3) above which a nucleus is pyknotic. Read the automatic cuts in viability_qc or the threshold figures first. Default None.",
"viability_negative_wells": "(list or str) - Untreated or vehicle wells, e.g. ['c1']; rows (r1), columns (c1) and wells (A01) all read. Their mean live-cell count is each plate's reference for the live-cell index, they read 0 on the cytotoxicity index and they are one side of its Z'. Blank leaves the index as the percentage of cells not live. Default None.",
"viability_positive_wells": "(list or str) - Wells given a cytotoxic control (for example digitonin, saponin or staurosporine), e.g. ['c12'], in the same notation as viability_negative_wells. They read 100 on the cytotoxicity index and, with the negative wells, give each plate's Z' in viability_qc; a Z' of 0.5 or more is a working assay. Blank scales the index to the negative wells alone. Default None.",
"viability_plate_map": "(str) - A table (CSV or Excel) of each well's compound and concentration: a well column (well such as A01, rowID and columnID, or prc), compound or treatment, concentration or dose, and optionally plateID. With it, viability, the cytotoxicity index and the infection of live cells are fitted against concentration per compound, and the host CC50 over the parasite EC50 is written as a selectivity index. Blank skips dose-response. Default blank.",
"cellprofiler_pipeline": "(str) - A CellProfiler pipeline (.cppipe) to run headless after measuring, in CellProfiler's own environment from the Model Zoo. Each field's channels are handed to it as <field>_ch<N>.tif, N from 0, and its masks as <field>_<object>_mask.tif; load the masks in NamesAndTypes as objects, and measure each object's Location. Its per-object measurements go to measurements.db:cellprofiler_<object>, matched to spaCR objects by prcfo. Blank skips it. Default blank.",
"bystander_measurements": "(bool) - Split uninfected cells into bystanders and distal cells. A bystander is an uninfected cell within the reach set by bystander_reach_in_diameters of an infected one; everything else uninfected is distal. Without this the two are the same row, so a bystander phenotype cannot be found and the uninfected control is a mixture of two populations whose variance hides the effect being looked for. Adds three columns per cell and costs one distance transform and one KD-tree per field. Default False.",
"bystander_reach_in_diameters": "(float) - How close an uninfected cell must be to an infected one to count as a bystander, expressed in measured cell diameters rather than pixels or micrometres, so it means the same thing at 20x and 63x. The diameter is the median of the cells in the field, ignoring those clipped by its edge. Zero or less makes every uninfected cell distal, which turns the split off without a second setting. Ignored unless bystander_measurements is enabled. Default 1.0.",
"spatial_measurements": "(bool) - Measure each object's neighbourhood: the number of neighbours within a radius, first and second nearest-neighbour distances, and the fraction of its border contacting another object. These measurements can be used to model density-associated variation in morphology and intensity. They are not produced for cytoplasm, which is defined as one object per cell. Computation requires one KD-tree and one boundary pass per field. Default True.",
"spatial_neighbor_radius": "(int) - Radius used by spatial_measurements when counting neighbouring objects. The value is expressed in the units recorded for the measurement table: pixels for two-dimensional data and micrometres for calibrated three-dimensional data. The radius is included in the output column name, so use one value consistently across plates that will be combined. Ignored unless spatial_measurements is enabled. Default 50.",
"radial_dist": "(bool) - Measure how each channel's intensity varies with distance from the nucleus, pathogen and organelle boundaries inside each cell, binned into 6 shells and saved as <object>_rad_dist_channel_<c>_bin_0-5. Keep it on to quantify recruitment or intensity gradients toward an object; turn it off to shrink the feature table and speed up measurement. Default True.",
"random_test": "(bool) - Seed the random draw of test-mode image sets with a fixed value (42), so every test run picks the same subset and results stay comparable. The selection is shuffled either way; set False when you want a different random subset each run to check that behaviour is not subset-specific. Default True.",
"randomize": "(bool) - Shuffle the order of the per-field arrays before they are grouped into normalization batches, so each batch spans plates and wells instead of one acquisition block - this matters because normalization percentiles are computed per batch. Forced to False for timelapse runs to keep frames in sequence. Default True.",
"regression_type": "(str) - Model family. Default 'mixed' nests guides inside genes as random effects, so guides disagreeing widens that gene's interval; it answers both levels at once, which is why 'level' greys out. Otherwise: 'ols', 'wls', 'rlm', 'huber' or 'quantile' for a continuous response; 'logit', 'probit', 'beta' or 'quasi_binomial' for fractions; 'poisson' for counts; 'lasso', 'ridge', 'elasticnet' or 'horseshoe' when the predictors outnumber the wells. Note regression_type 'beta' fits a beta GLM, while transform 'beta' only transforms the response.",
"regression_panel_manifest": "(dict, str, or None) - Publication panels built after the regression CSVs are written. None skips them. Supply a mapping, or a path to JSON/YAML, declaring each panel's result source, phenotype, gene/gRNA level and plot kind; the packages land under publication_panels/. Structured rather than a one-line field, so it is set from a script or a saved settings file rather than in the GUI. Default None.",
"remove_background": "(bool) - Hard-clip every pixel below the 'background' value to zero before normalization and segmentation. Use it when a channel carries a bright, even haze that inflates the normalization floor; leave it off for dim or already flat-fielded data, since the clip silently deletes faint real signal. Default False.",
"remove_background_cell": "(bool) - Before normalisation, zero every pixel in the cell channel below cell_background. This flattens haze so the percentile stretch is driven by real signal, but it also erases genuinely dim cell edges and can shrink masks. Enable only once cell_background is set from an actual empty region. Default False.",
"remove_background_nucleus": "(bool) - Before normalizing the nucleus channel, zero every pixel below nucleus_background and exclude those pixels from the percentile calculation. Enabling it raises contrast on real nuclei and suppresses haze, but clips genuinely dim nuclei to zero so they may become unsegmentable. Default False; check nucleus_background against raw images first.",
"remove_background_organelle": "(bool) - Before normalising the organelle channel, hard-zero every pixel whose raw intensity is below organelle_background. Enable it when diffuse autofluorescence inflates the low percentile and faint puncta are lost in haze; leave it off for dim organelles, since the clipping erases real signal and biases downstream intensity measurements. Default False.",
"organelle_background": "(int or float) - Raw intensity treated as background on the organelle channel: the floor that remove_background_organelle clips to zero, and the base of the organelle_signal_to_noise product. Read it off a blank region of the image rather than guessing. Default 100.",
"organelle_signal_to_noise": "(int or float) - Multiple of organelle_background an organelle pixel must reach to count as signal during normalisation. Raise it when background haze is being normalised as if it were organelle; lower it when faint puncta are being flattened away. Default 10.",
"remove_background_pathogen": "(bool) - Before normalising the pathogen channel, hard-zero every pixel whose raw intensity is below pathogen_background. Enable it when diffuse autofluorescence inflates the low percentile and Cellpose starts segmenting haze; leave it off for dim parasites, since the clipping erases real signal and biases downstream intensity measurements. Default True.",
"remove_cluster_noise": "(bool) - Remove points that DBSCAN labels as noise (-1) before plotting the embedding, so the figure contains only clustered points. Disable it to retain all embedded points, including diffuse background. It has no effect with kmeans, which never emits -1, and is disabled automatically when color_by is set. Default True.",
"remove_highly_correlated": "(bool or float) - Before dimensionality reduction, drop numeric features whose absolute Pearson correlation with an already-kept feature exceeds a cut-off. Pass a float to set the cut-off yourself, True to use 0.95, or False to keep everything. Enable it so families of near-duplicate measurements (area, perimeter, convex_area) do not dominate the embedding. Default True.",
"remove_highly_correlated_features": "(bool) - In the machine-learning feature table, drop any feature whose absolute Pearson correlation with an already-kept feature exceeds 0.95, applied after the channel_of_interest filter. Leave it on so redundant measurements do not split importance scores and slow fitting; turn it off only when you need every original column. Default True. Note the UMAP path uses remove_highly_correlated instead.",
"remove_image_canvas": "(bool) - When object thumbnails are overlaid on the embedding plot, make zero-valued background pixels transparent so only the segmented object is visible. Enable it to remove black thumbnail backgrounds, especially with black_background. Only L, I and RGB crops are supported; other PIL modes raise an error. Default False.",
"remove_low_variance_features": "(bool) - Drop numeric features whose variance across objects falls below 0.01 before model fitting -- near-constant columns that carry no discriminative signal but still cost time and dilute importance rankings. Turn it off only when your features live on a very small numeric scale, where genuine signal can fall under that fixed cut-off. Default True.",
"regression_backend": "(str) - Selects the library and device that fit the chosen regression_type. 'statsmodels (CPU)' preserves the established results and is the default. 'torch (GPU)' accelerates mixed models, 'pyfixest (CPU)' accelerates OLS/WLS with absorbed plate-position effects, and 'glum (CPU)' accelerates wide GLMs. The selector describes measured costs and numerical differences for each option; unavailable or incompatible backends are greyed out with the reason. Default 'statsmodels (CPU)'.",
"l1_ratio": "(float) - How the elastic-net penalty is split between L1 and L2: 1.0 is a pure lasso (sparse, picks one gRNA out of a correlated group), 0.0 is a pure ridge (dense, shares the effect across the group), and values between keep some of both. Use 0.5 when correlated gRNAs of the same gene should be selected together rather than arbitrarily. Read only by regression_type 'elasticnet'. Default 0.5.",
"quantile": "(float) - Which quantile of the response quantile regression fits, strictly inside 0 and 1: 0.5 is the median (robust to outlier wells), 0.9 asks which gRNAs move the top of the distribution rather than its centre. Aggregation is turned off automatically so the quantile is taken over cells, not over well means. Read only by regression_type 'quantile'; it replaced the old overload of alpha. Default 0.5.",
"hinge_threshold": "(float) - Response value above which a well counts as positive for the hinge (linear SVM) fit. Leave it None when the response is already binary, in which case the two values it holds become the two classes. spaCR refuses a continuous response with no threshold rather than splitting it at the mean or median, because a cut chosen by the software decides the hypothesis being tested. Read only by regression_type 'hinge'. Default None.",
"hinge_n_boot": "(int) - Number of bootstrap resamples behind the hinge p-values. A support vector machine has no likelihood and so no Wald test; spaCR refits it on this many resamples of the wells and compares each coefficient to its bootstrap standard deviation. Treat the result as a stability statistic, not a hypothesis test. Higher is steadier and linearly slower; below about 50 the standard deviations are too noisy to rank on. Default 200.",
"spline_knots": "(int) - How many knots the spline basis gets for each CONTINUOUS covariate. More knots let the covariate bend more freely and spend more degrees of freedom; a covariate with fewer distinct values than the degree is left linear rather than given a basis made out of nothing. The guide columns are never given a basis, so the fit still returns one coefficient and one p-value per guide. Read only by regression_type 'spline'. Default 4.",
"spline_degree": "(int) - The polynomial degree of each covariate's spline basis. 3 is a cubic spline, the usual choice; 1 is piecewise linear. Read only by regression_type 'spline'. Default 3.",
"huber_t": "(float) - Where Huber's loss switches from squared to linear, in units of the estimated residual scale, for the robust fits. Smaller values downweight more wells and resist heavier contamination; larger values approach ordinary least squares. The default 1.345 gives 95 percent of the efficiency of OLS under normally distributed residuals. Read only by regression_type 'rlm' and 'huber'. Default 1.345.",
"lasso_n_boot": "(int) - Number of bootstrap resamples used to rank lasso and elastic-net hits by how often each gRNA survives the penalty. These models have no valid p-values, so selection frequency replaces the significance test entirely. Higher is steadier and linearly slower; the cost is one full penalised fit per resample, doubled when alpha is 'auto' because each resample cross-validates. Default 200.",
"group_lasso_lambda": "(float) - Penalty weight of the group lasso, which shrinks all of one gene's guides together rather than one at a time, so a gene enters or leaves the model as a unit instead of on its luckiest guide. Larger values keep fewer genes; 0 leaves the fit unpenalised and negative is refused. Set it to auto to choose it by cross-validation. Default auto.",
"lasso_selection_threshold": "(float) - Minimum bootstrap selection frequency, between 0 and 1, for a lasso or elastic-net coefficient to be called a hit. 0.6 means the gRNA kept a non-zero coefficient in at least three fifths of the resamples. Raise it for a shorter, harder-to-argue-with list; lowering it below about 0.5 admits terms the penalty drops as often as it keeps. Default 0.6.",
"regression_qc": "(bool) - Write variance-homogeneity, residual, design, influence, and calibration diagnostics to <res_folder>/regression_qc/ as figures, a combined PDF, and a text report. One fit requires approximately 5.8 seconds and writes 19 files, so this is enabled for individual analyses but disabled automatically during parameter sweeps to avoid producing thousands of diagnostic files. Reopen a selected trial to generate its diagnostics. Applies to every regression_type. Default True.",
"model_plate_position": "(bool) - Include plateID, rowID, and columnID terms to model plate and spatial effects. random_row_column_effects selects fixed terms or a plate-grouped mixed model with row/column variance components. Enabling random effects requires this setting. In the reference screen the layout terms were jointly significant at p = 6.7e-23, so enable them when plate, edge, row, or column effects are plausible. Default False.",
"random_row_column_effects": "(bool) - Fit plate, row and column as random effects instead of fixed ones: True overrides regression_type to 'mixed' and fits a MixedLM grouped by plateID with rowID and columnID variance components, dropping them from the fixed-effect formula. Use it when edge or row artefacts differ between plates; it is slower and may fail to converge. Default False.",
"resample": "(bool) - Passed to Cellpose model.eval: run the mask-tracking dynamics at full image resolution instead of on the downsampled network grid. Enabling it gives smoother, better-fitting object outlines at the cost of time and memory, and helps most when objects differ a lot from the model's training diameter. Default False; the object pipeline sets True for cell/nucleus and False for pathogen.",
"rescale": "(bool) - Let Cellpose rescale each image by 30/diameter before segmenting, so objects arrive at the size the model expects. Turn off only when the diameter is already correct for the model. Default False.",
"resnet_features": "(bool) - Placeholder for embedding raw crops with ResNet features instead of the measured feature table. The branch in generate_image_umap is an empty pass, so enabling it skips the embedding step entirely and the run then fails on an unbound 'embedding' variable. Leave it False. Default False.",
"row_limit": "(int) - Randomly subsample the joined measurement table down to this many objects (fixed seed 42) before dimensionality reduction, keeping UMAP and clustering tractable. Raise it for a more faithful map at higher memory and runtime cost, or set to None to use every row. Must not exceed the available row count. Default 1000.",
"save_arrays": "(bool) - Also save each object as a raw .npy array - all channels, cropped to its bounding box, unnormalised - under a region_array/ folder. Enable when you need full bit depth or channels beyond png_dims for custom analysis; it uses far more disk than PNGs. Requires save_png to be True as well. Default False.",
"save_figure": "(bool) - Write the embedding, plus the cluster grid when plot_cluster_grids is on, as vector PDFs to <src>/results/<METHOD>_embedding.pdf and <METHOD>_grid.pdf. Enable it when you want the figure for a paper or a record of the run; either way the plots are still displayed on screen. Default False.",
"save_measurements": "(bool) - Master switch for the measurement half of measure_crop: compute morphology and intensity features for every cell, nucleus, pathogen, organelle and cytoplasm object and write them to the plate's SQLite database. Set it False when you only want cropped PNGs or filtered masks -- segmentation and cropping still run, but no measurement tables are written. Default True.",
"save_png": "(bool) - Write one PNG crop per segmented object into <crop_mode>_png/ and register each path in the png_list table of measurements.db. Required for training or applying a classifier, for the Annotate app and for the UMAP image plots. Turn off to only compute measurements and save time and disk. Default True.",
"Signal_to_noise": "(int) - Background multiplier used as the Cellpose normalization threshold (background * Signal_to_noise). Per channel, spaCR selects the first of the 98th, 99th, 99.9th, 99.99th and 99.999th percentiles above it and rescales to that value. Higher values reduce clipping and make output dimmer; lower values reveal faint signal but can saturate bright objects. If no percentile qualifies, the range collapses to the 2nd percentile, indicating that this value is too high. Ignored when percentiles is set. Default 10 (5 in check_cellpose_models).",
"smooth_lines": "(bool) - Draw cluster outlines as a smoothed spline through the convex hull (2 pt wide) rather than the raw straight hull segments (4 pt). Purely cosmetic - it does not change clustering; switch it off if smoothing distorts the true cluster boundary. No effect unless plot_outlines is on, and forced off when color_by is set. Default True.",
"src": '(str, path) - Folder the current step reads from and writes into: raw images for mask generation, the merged/ folder of .npy stacks for measure, the plate root for dataset/regression steps, or the folder of .fastq.gz reads for sequencing. Outputs (stack/, masks/, measurements/measurements.db, datasets/, results/) are created inside it. A list of paths, or a "[\'a\',\'b\']" string, processes several plates in one run. No usable default: the settings factories fill a placeholder (\'path\' or \'/path/to/src\'), so this must be supplied.',
"target": "(str) - Free-text label for the protein or marker imaged in channel_of_interest, e.g. 'GRA1'. The recruitment run prints it in its banner ('channel:3 = protein') to record what the recruitment ratio is measuring; it feeds no computation, so changing it alters nothing but that log line. Default 'protein'.",
"target_height": "(int) - Height in pixels that images are resized to before segmentation; masks are scaled back to the original dimensions afterwards. Only applied when both target_height and target_width are set (and, on the non-normalized path, when resize is True). Use it to match the field size the model was trained at. Default None, which disables resizing; 1120 for plaque analysis.",
"target_intensity_min": "(float) - Recruitment-analysis cutoff on the 95th-percentile intensity of channel_of_interest inside each cell: cells at or below it are discarded before recruitment ratios are computed. Raise it to keep only strongly expressing cells; set 0 or None to disable the filter entirely. Raw intensity units, default 1.",
"target_width": "(int) - Width in pixels that images are resized to before segmentation; masks are scaled back to the original dimensions afterwards. Only applied when both target_width and target_height are set (and, on the non-normalized path, when resize is True). Use it to match the field size the model was trained at. Default None, which disables resizing; 1120 for plaque analysis.",
"tables": "(list) - Measurement tables read from each plate's database and merged into one analysis frame. Only 'cell', 'nucleus', 'pathogen', 'cytoplasm', and 'png_list' are merged. Any other table, including 'organelle', is loaded but omitted from the merged result without a warning. Default ['cell', 'nucleus', 'pathogen', 'cytoplasm'].",
"test": "(bool) - In classifier training, run the held-out evaluation pass, either with train or alone to score an existing model. In the sequencing barcode mapper, process only the first read chunk and print a preview so the regex and barcode CSVs can be validated quickly. Default False.",
"test_images": "(int) - How many plate/well/field image sets are copied into a test/ folder when test_mode is on; every channel file belonging to a chosen set is copied together. Raise it for a broader smoke test, lower it for a faster one. Forced to 1 for timelapse runs so a full sequence stays intact. Default 10.",
"test_mode": "(bool) - Run the pipeline on a small random subset instead of the whole folder. Mask generation copies test_images (default 10) complete image sets into <src>/test and works there; measure_crop copies test_nr (default 10) merged arrays into test/merged. Both also force verbose and plot on. Use it to check channel assignment, diameters and thresholds before committing to a full plate. Default False.",
"dry_run": "(bool) - Validate settings against the selected data, report the planned operations, and stop before any compute begins. Checks that src contains the expected files, channel and mask-plane indices are valid, and required models, barcode CSV files, or measurements.db files are present. Each problem is reported with a suggested correction, followed by a summary of the planned segmentation, measurement, and output locations. Nothing is written and no model is loaded. Default False.",
"test_nr": "(int) - How many files are sampled at random from merged/ into test/merged when test_mode is on in the measure-and-crop pipeline, so measurement runs on a small subset. Raise it if a handful of fields is not representative; each extra file costs a full measurement pass. Default 10.",
"treatment_loc": "(list of lists) - Plate wells that received each entry of treatments, one inner list per treatment in the same order, e.g. [['r1','r2'],['r3']]. Identifiers must start with 'r' (row) or 'c' (column); wells you do not list get no treatment label. Used by the vision-score annotation step. No default - supply it alongside treatments.",
"treatments": "(list or None) - Names of the drug or treatment conditions in the experiment, e.g. ['dmso','lovastatin']. Each name is written into the treatment column and folded into the combined condition label used for grouping and plotting; positionally paired with treatment_plate_metadata (or treatment_loc), which lists the wells for each. Default ['cm','lovastatin']. Invasion and Replication start at None, so they add no treatment condition until names and matching well groups are supplied.",
"top_features": "(int) - Feature cap in the ML screen analysis: how many rows the feature-importance and permutation-importance bar plots show, and how many top-ranked features the SHAP refit and its summary plot use. It is also the k of the SelectKBest pruning applied before the model is fitted, but only when prune_features is True - with prune_features at its default False the classifier trains on every feature and this is reporting/SHAP scope only. Raise for a fuller picture, lower for readable plots. Default 30.",
"train": "(bool) - Run the training stage. Disable this setting to apply an existing model_path to a dataset without retraining, as when scoring a new plate with a previously trained model. Default True.",
"intercept": "(str) - Definition of the fitted intercept. 'fitted' estimates it from the data as the predicted response when every predictor is at its reference level; in a one-hot gene design, this reference is the gene omitted by patsy and may not have a useful biological interpretation. 'control' subtracts the negative controls' mean response before fitting and suppresses the intercept term, so every coefficient is a difference from the controls. 'zero' fits through the origin. 'value' fixes the intercept at intercept_value. Default 'fitted'.",
"intercept_value": "(float) - The number the intercept is pinned to when intercept is 'value'. The response is shifted by it and the term suppressed, so the fit is exactly y = intercept_value + terms. Read for no other intercept setting, and the panel greys it out for them. Default 0.0.",
"transform": "(str or None) - Optional transform applied to the aggregated per-well response before fitting: 'log' (log1p), 'sqrt', 'square', 'beta' (logit for a proportional response, with endpoints moved away from 0 and 1 and reported in the summary), or None. Use a transform when the response is skewed and fails the normality check. The fit then reports coefficients for '<transform>_<dependent_variable>'. Default None. Regression starts at 'log', so its first fit applies log1p unless this is changed.",
"val_split": "(float) - Fraction of src/train randomly held out as a validation set in each run (0.1 = 10 percent). The validation score controls checkpoint selection, early stopping and live training curves; at 0 there is no validation loader, so checkpointing uses training accuracy and may favour memorisation. Increase it on small datasets for a less variable estimate. When a grouping level is set, complete groups are held out, so the realised fraction is quantised and may differ substantially from the requested value; both values are reported. Default 0.1.",
"verbose": "(bool) - Print the resolved settings table, channel and model choices per object type, row counts per table, and object counts after each filter. It only adds console output; enable it to identify which stage produced an unexpected object count. The default is True for mask, UMAP, screen analysis, barcode mapping and Cellpose training, and False for measure, plotting helpers and regression. Invasion and Replication also start with console detail disabled.",
"weight_decay": "(float) - L2 penalty applied to the weights on every optimizer step (AdamW applies it decoupled from the gradient). Raise it, toward 1e-3 to 1e-2, when validation loss climbs while training loss keeps falling; lower it toward 0 when the model cannot fit the training set at all. Every supported optimizer honours it. Default 0.00001.",
"width_height": "(list of int) - Legacy Cellpose checkpoint metadata retained for older settings files. Current training resizes every image-mask pair with the scalar target_size and does not read this list, so changing it does not change training. Default [1000, 1000].",
"file_type": "(str) - Image FORMAT the pre-generated crops are in, as a file extension: 'png', 'tif', 'tiff', 'jpg', 'jpeg', 'bmp' or 'npy'. It is a format filter and nothing else - WHICH OBJECT a crop is of is path_string's job, so the pair can express 'every nucleus crop, whatever format' and 'every TIFF, whatever object', which one combined setting never could. A legacy value of the form '<object>_png' is still accepted and read as its extension, so an old settings file keeps working. Blank accepts any format. Default 'cell_png', which is read as 'png'.",
"model_path": "(str) - Path to a trained spaCR classifier saved as a whole PyTorch object (loaded with torch.load(weights_only=False), not a state_dict). Used when applying a model to a dataset tar and when generating activation maps. deep_spacr overwrites it with the freshly trained model whenever train is True, so set it only to score with an existing model. Default ''.",
"dataset": "(str) - Path to the .tar archive of single-object PNG crops produced by generate_dataset, which the activation-map step opens with TarImageDataset. The plate folder is inferred two levels above it and CAM outputs are written next to it under <tar_name>/<cam_type>/. Provide an absolute or directory-qualified path rather than a filename alone. Default ''.",
"score_threshold": "(float) - Probability cutoff (0-1) applied to the model's positive-class score when deriving binary predictions. With test-time augmentation, this threshold labels each view before agreement and voting are calculated. Probability scores remain available. Default 0.5.",
'tta_enabled': '(bool) - Score selected rotated or reflected versions of each phenotype crop during inference. Choose transforms in this category; the original view is always included. Saves original predictions, mean probabilities, standard deviations, agreement and review flags. Keep disabled when orientation carries biological meaning. Does not change training augmentation. Default False.',
'tta_rotations': '(bool) - With test-time augmentation enabled, include 90, 180 and 270 degree rotations. Uses exact pixel rotations without interpolation. Default False.',
'tta_horizontal_flip': '(bool) - With test-time augmentation enabled, include horizontal reflections of the selected rotations. Each distinct reflected view is classified before aggregation; equivalent orientations are evaluated once. Disable when horizontal direction has biological meaning in the assay. Default False.',
'tta_vertical_flip': '(bool) - With test-time augmentation enabled, include vertical reflections. Equivalent orientations are evaluated once; all rotations and both flips produce eight distinct views. Default False.',
'tta_aggregation': '(str) - Combine test-time views using probability_mean or majority_vote. Voting ties prefer the higher mean probability, then the lower class index. pred retains the mean probability; predicted_label and cv_predictions retain the selected label. Default probability_mean.',
'tta_min_agreement': '(float) - Flag an object for review when fewer than this fraction of orientation predictions agree with the selected label. Agreement measures orientation stability, not calibrated confidence. Range 0-1; default 0.75.',
'tta_max_std': '(float) - Flag an object when its positive-class probability (binary) or selected-class probability (multiclass) has a population standard deviation above this value across orientations. Range 0-1; default 0.15.',
"sample": "(int, list or None) - Randomly draw this many PNG crops from the database when building the dataset tar instead of using all of them; a list uses its first element, and values above the total are clamped. Use it to build a quick trial dataset or to cap a huge screen. None uses every crop, shuffled. Default None.",
"file_metadata": "(str, list, or None) - Substring filter applied to png_path when retrieving crops from the database. Only paths containing the supplied string are included; a list matches any entry rather than requiring all entries. Use this setting to restrict a dataset to one plate, well, or object type, for example 'plate1_' or 'cell_png'. None includes every crop. Default None.",
"apply_model_to_dataset": "(bool) - After training (or straight away when reusing a saved model_path), pack the object PNGs into a tar, run inference over it, copy the n_top_examples most confident images per class into top_examples/, and merge the per-object scores back into measurements.db. Turn it off to only train and evaluate a model without scoring the screen. Default True.",
"generate_full_dataset": "(bool) - Build the full unlabelled inference dataset tar from every selected plate independently of training or model application. Apply model to dataset also creates it automatically when needed. API: spacr.io.generate_dataset. Default False.",
"tar_path": "(str) - Existing full-dataset tar to reuse for inference. Leave blank to generate one beneath the first plate's datasets folder. Multiple selected plates are combined into one tar. API: spacr.deep_spacr.apply_model_to_tar. Default empty.",
"n_top_examples": "(int) - Number of highest-confidence images saved per predicted class after full-dataset inference. This gives a quick visual check of class meaning and common errors. Default 20.",
"random_seed": "(int) - Reproducibility seed shared by labelled train/test splitting, train/validation splitting, and grouped cross-validation folds. Keep it fixed to reproduce a run; vary it to evaluate sensitivity to the sampled partition. Default 42.",
"balance_to_smallest": "(bool) - Downsample every generated training class to the size of the smallest class before writing train/test folders. This removes the dataset prior but discards majority examples; disable it and use class_balance during training to retain all images. Default True.",
"write_random_annotation_column": "(bool) - In annotation mode, persist an automatically selected unannotated comparison group into png_list as <column>_random. This makes an automatically generated control class reproducible and auditable. Default False.",
"train_channels": "(list) - Which colour planes of each object crop the classifier sees, chosen from 'r', 'g' and 'b'. Fewer channels means a smaller input tensor and a model that cannot use the dropped stain, so drop a channel only when it carries no signal for your phenotype. The joined letters also become part of the saved model's filename. Default ['r', 'g', 'b'].",
"classifier_family": "(str) - Which classifier the merged Classify module runs: 'cv' trains a Torch model on object crops, 'ml' fits a gradient-boosted model on measured features. The settings the other family uses are greyed out, not hidden - they keep their values. This is the only Classify screen; the two it replaced were removed on 2026-08-23, and their entry points (spacr.deep_spacr.deep_spacr and spacr.ml.generate_ml_scores) are what this dispatches to unchanged. Default 'cv'.",
"dataset_mode": "(str) - How training classes are defined: 'metadata' splits crops by well metadata, 'annotation' by the values in one or more annotation columns of png_list. Either way the classes themselves are set in the Classes editor, which names a column and a value per class. A settings file written before the 'measurement' basis was removed still loads: it is read as 'annotation', which is what its threshold rules resolved to after writing their label column. Any other value aborts and returns no dataset. Default 'metadata'.",
"annotated_classes": "(list) - Legacy setting that is not read from a module configuration. It remains available only as a direct parameter of io.training_dataset_from_annotation and io.training_dataset_from_annotation_metadata. Define classes with the Classes editor instead: select metadata values in metadata mode, or annotation columns and values in annotation mode. Default [1, 2], with no effect in module settings.",
"um_per_pixel": "(float) - Physical size of one image pixel in micrometres, determined by the acquisition objective and camera. It is used only to convert scale_bar_length_um into pixels when drawing a scale bar on representative-image grids; an incorrect value produces an inaccurate scale bar but does not rescale or resample the images. The plotting helpers default to 0.1.",
"pathogen_model": "(str or None) - Path to a custom Cellpose checkpoint used to detect pathogen objects, overriding pathogen_model_name when set. It must be a CPSAM-architecture checkpoint (one your own Train Cellpose run produced); a Cellpose-3 CPnet file cannot load into Cellpose 4. A path that does not exist stops the run rather than falling back to the stock weights silently. Default None.",
"timelapse_displacement": "(int or None) - Maximum distance in pixels an object may travel between consecutive frames when linking: trackpy's search_range, or btrack's max search radius. Too small fragments tracks, too large causes identity swaps and SubnetOversize failures. None auto-searches downward from 500 for trackpy and falls back to 100 for btrack. Default None.",
"timelapse_memory": "(int) - Number of consecutive frames an object may vanish (e.g. missed by segmentation) and still be re-linked to the same track by trackpy. Raise it when tracks fragment because objects blink out; too high risks merging two different objects into one track. Not used by the btrack mode. Default 3.",
"timelapse_mode": "(str) - Tracking backend used to link objects between frames. 'trackastra' is a pretrained transformer with division-aware linking; 'ultrack' jointly optimizes segmentation and linking and supports dense or 3D data at a higher computational cost; 'trackpy' uses a configurable search radius and frame memory; 'btrack' uses a motion model; 'iou' links masks by overlap and may fail when objects move far between frames; 'timeflows' is spaCR's experimental temporal Cellpose, needs a trained checkpoint in timeflows_model and has not yet beaten 'iou' on held-out movies. Default 'trackastra'.",
"trackastra_model": "(str) - Pretrained Trackastra checkpoint used for frame linking. 'general_2d' is the general-purpose two-dimensional model for live-cell data. This setting is used only when timelapse_mode='trackastra'. Select a different checkpoint only when it was trained for substantially different image characteristics. Default 'general_2d'.",
"trackastra_linking": "(str) - How Trackastra turns predicted association scores into tracks: 'greedy' takes the best match per object and is fast, 'ilp' solves the assignment globally and is more accurate on crowded or dividing populations but needs the trackastra ilp extra and considerably more time. Default 'greedy'.",
"ultrack_max_distance": "(float) - The largest jump in pixels Ultrack will consider when linking an object in one frame to a candidate in the next; anything further apart is never joined, so the track breaks instead. Raise it for fast-moving or sparsely sampled cells, lower it on crowded fields where a generous radius invites identity swaps. Only consulted when timelapse_mode='ultrack'. Default 25.0.",
"ultrack_division_weight": "(float) - Cost the Ultrack solver pays to split one track into two daughters; the value is negative and the more negative it is the more readily divisions are accepted. Make it less negative when a replication assay over-calls divisions on touching cells, more negative when real division events are being missed. Only consulted when timelapse_mode='ultrack'. Default -0.1.",
"ultrack_contour_sigma": "(float) - Standard deviation of the Gaussian blur applied while turning the segmentation labels into the contour map Ultrack builds its candidate objects from. Zero keeps the boundaries exactly as Cellpose drew them; one to four softens them so the joint solver is free to redraw boundaries between objects that were merged or split. Only consulted when timelapse_mode='ultrack'. Default 0.0.",
"timeflows_model": "(str) - Path to a trained Timeflows checkpoint, spaCR's experimental temporal Cellpose that predicts where each object's centre moves in the next frame and links masks from those predictions. Only consulted when timelapse_mode='timeflows', which is never the default: on held-out movies measured so far it does not beat overlap linking ('iou'). No checkpoint is downloaded; leave empty unless you trained one. Default None.",
"ultrack_n_workers": "(int) - How many worker processes Ultrack runs during its candidate-segmentation and linking passes; they all write into the same temporary sqlite store, so extra workers cut wall-clock on long movies but add database contention and memory. Leave it at one for short batches or a busy machine. Only consulted when timelapse_mode='ultrack'. Default 1.",
"timelapse_frame_limits": "(list) - Slice of frame indices [start, end] kept from each batch before tracking, e.g. [0,10] to work on the first ten frames while tuning settings. The list is ignored unless it has at least two elements, which is why the shipped default [5,] has no effect. Default [5,].",
"timelapse_objects": "(list) - Which segmented objects are tracked across frames and relabelled with track IDs: any subset of ['cell', 'nucleus', 'pathogen']; any other value aborts the run with a message. Each extra entry costs a full additional tracking pass. Tracking nuclei is often more stable than cells when cells touch. Default ['cell'].",
"timelapse_lineage": "(bool) - Build lineage trees after tracking, preserving native division links and recording the actual frame filenames and final object labels. Measure can rebuild trees using a numeric measured feature, with outputs in tracks/lineage_measured; original tracks, measurements and pre-Measure lineage outputs remain unchanged. Measured colours require the saved frame/source mapping; rerun tracking with lineage enabled if it is missing. Frame quantities are retained, with hours added only when time_s or frame_interval_s supplies calibration. Default False.",
"timelapse_lineage_color_by": "(str) - Colour each lineage segment by generation_time (frames), generation_time_hours, generation, start_frame or n_frames, or a numeric feature averaged over its observed frames. Tracking reads tracks-table columns; Measure reads the selected tracked object's measurement table using saved frame and label identities. Means use available measured frames; segments with no measured values remain uncoloured. Ambiguous identities or invalid features stop that field's measured-colour export. Default generation_time.",
"timelapse_lineage_max_distance": "(float) - Largest distance in pixels between a mother's last position and a new track's first position for the new track to count as her daughter when divisions are inferred. Raise it for large cells or long frame intervals, lower it when neighbours are wrongly joined. Not used for division links the tracker reports. Ignored unless timelapse_lineage. Default 30.0.",
"timelapse_lineage_min_division_h": "(float) - Shortest plausible time in hours between two divisions of one cell; 6 suits Toxoplasma endodyogeny. With a frame time (time_s or frame_interval_s), no division is inferred in a movie shorter than half of it, nor from a cell born less than this long before. Inferred daughters must also persist for 3 frames, and a track ending at the field edge or after a gap counts as leaving the field. Blank turns the time limit off. Ignored unless timelapse_lineage. Default 6.0.",
"timelapse_events": "(bool) - After the run, detect events on every tracked object with a small neural network that reads short windows of each track (shape, intensity, movement, tracks starting or ending nearby, and image crops): mitosis, egress, invasion, host death or whatever classes timelapse_events_annotations names. Writes time-stamped events, lineage trees re-linked from detected mitoses and Kaplan-Meier time to each event per condition to tracks/events. The small encoder runs on CPU; the optional videomae encoder uses its separately installed worker and declared video device. Default False.",
"timelapse_events_annotations": "(str or None) - Table of hand-annotated events with columns field (the tracks file's field name), track_id, frame and event, plus an optional object column. Every event of an annotated field must be listed. The detector is scored on held-out annotated fields (precision, recall and timing error in frames, matched within 2 frames), trained on all of them and saved as tracks/events/event_model.pt. Blank uses timelapse_events_model. Default None.",
"timelapse_events_model": "(str or None) - A trained event model (event_model.pt from an earlier run) to apply when timelapse_events_annotations is blank. It is read as tensors only. Default None.",
"timelapse_events_window": "(int) - Frames the detector sees around each frame of a track, centred on it; an even number is raised by one. Longer windows see slower changes but blur events close together. Only read when training. Default 9.",
"timelapse_events_threshold": "(float) - Smallest class probability, from 0 to 1, at which a frame is called an event; each event keeps the most probable frame within 2 frames. Raise it for fewer false events, lower it to miss fewer. Default 0.5.",
"timelapse_events_conditions": "(list or None) - Conditions compared in the event timing, as name=wells entries in the plate-map notation, such as ['mock=c1,c2', 'drug=c3,c4']; fields in wells no entry names are left out. Blank compares wells. Default None.",
"timelapse_events_encoder": "(str) - small retains the original CPU crop encoder. videomae uses frozen pretrained video features from its separately installed Model Zoo backend, combined with track features in a classifier trained on the experiment's event annotations. Its Kinetics labels are not biological event labels. Held-out microscopy evaluation remains required. Default small.",
"timelapse_events_video_checkpoint": "(str or None) - Local folder containing the pinned official VideoMAE config.json, preprocessor_config.json and model.safetensors. Every SHA-256 is checked before and after inference; no weights are downloaded. The official checkpoint is CC-BY-NC-4.0. Required when training with videomae; a saved model can reuse its recorded folder. Default None.",
"timelapse_events_video_channels": "(list or None) - Exactly three ordered source-channel indices for VideoMAE's RGB input, such as [0,1,2] or explicitly [0,0,0] for monochrome. Crop intensities normalized by the movie's channel percentiles are clipped to [0,1], sampled at 16 evenly spaced frames and processed with the pinned checkpoint's image processor. Mapping and preprocessing identity are saved with the event model. Default None.",
"timelapse_events_video_device": "(str) - auto selects CUDA, then Metal, then CPU inside VideoMAE's separate worker; cpu forces CPU feature extraction. The frozen encoder processes one clip at a time and the small event classifier trains on CPU. Default auto.",
"timelapse_remove_transient": "(bool) - After linking, drop every track not present in all frames (trackpy filter_stubs over the full stack length), keeping only objects tracked from first frame to last. Enable for clean per-object time courses; expect to lose cells that divide, enter or leave the field, so object counts fall. Default False.",
"timelapse": "(bool) - Treat each well/field as a time series instead of independent images: files are grouped into time stacks, randomization is switched off, per-channel movies are written, objects in timelapse_objects are tracked across frames, a timeID column is added to the measurement tables, and measure_crop stops writing single-object PNGs. Only enable when filenames carry a time index. Default False.",
"pathogen_min_size": "(int) - (Deprecated) Minimum pathogen object area in pixels squared, applied during measurement: any label with fewer pixels than this is erased from the pathogen mask before features are extracted. 0, the default, disables it. Superseded by an 'area' row for pathogen in object_filters, which filters at segmentation time instead.",
"pathogen_mask_dim": "(int) - Position along the last axis of each merged/*.npy array where the pathogen label mask sits, one plane after the nucleus mask. With the default four image channels (0-3) that is 6; shift it if you keep a different number of channels. None makes measure_crop skip pathogen measurements, so infection status cannot be scored. Default 6.",
"use_bounding_box": "(bool) - Crop the object's rectangular bounding box padded by 10 px instead of its mask, so neighbouring cells and background inside the box are kept rather than zeroed out. Enable when the classifier should see local context; leave off to isolate a single object on a black background. Default False.",
"plot_points": "(bool) - Show the scatter marker for each object in the embedding. When False the markers are still drawn but at alpha 0, so cluster colors and the legend survive while only the outlines and overlaid thumbnails stay visible - handy for image-only UMAP figures. Marker size comes from dot_size. Default True.",
"pos": "(str) - Column ID marking positive-control wells in the image UMAP. Rows whose columnID equals it are labelled cond='pos', so exclude_conditions can drop them; and when embedding_by_controls is True the rows whose col_to_compare equals it help fit the reducer. Default 'c1' (note: not 'c2').",
"neg": "(str) - Column ID marking negative-control wells in the image UMAP. Rows whose columnID equals it are labelled cond='neg', so exclude_conditions can drop them; and when embedding_by_controls is True the rows whose col_to_compare equals it join pos in fitting the reducer. Default 'c2' (note: not 'c1').",
"pathogen_plate_metadata": "(list of lists) - Well locations of each pathogen condition, one inner list per entry in pathogen_types. Every item must be a row or column ID string such as 'c1' or 'r3'; anything else is silently ignored and those wells stay unannotated. Ranges like 'c2-c11' are not expanded - list each row/column. Do not leave it None while pathogen_types is set: annotation is not skipped, every row is labelled with the first pathogen_types entry. Defaults: None in the plot-from-db settings, [['c1','c2','c3'],['c4','c5','c6']] for recruitment analysis.",
"treatment_plate_metadata": "(list of lists) - Wells that received each treatment, with one inner list per treatment in the same order, for example [['r1','r2','r3'],['r4','r5','r6']]. Entries must start with 'r' (row) or 'c' (column); other entries are ignored and receive no treatment label. Unlisted wells remain in the output, and their condition values contain only the available cell, pathogen, or treatment labels. Default None. Recruitment starts with [['r1', 'r2', 'r3'], ['r4', 'r5', 'r6']], positionally paired with its two initial treatment names.",
"regex": "(str) - Regex applied with re.match to each extracted read window; it must define the named groups columnID, grna and rowID, whose captured sequences are looked up in the three barcode CSVs. Non-matching reads are silently dropped, so a wrong group name or barcode orientation yields zero counts. The default captures an 8 bp column, 20-21 bp gRNA and 8 bp row barcode.",
"target_sequence": "(str) - Constant vector sequence used as the anchor: every read is scanned for an exact match and the barcode window is then sliced relative to that hit using offset_start and window_length. Reads without an exact match are skipped entirely, so it must be error-free and given in the orientation of the read being scanned. Default 'TGCTGTTTCCAGCATAGCTCTTAAAC'.",
"barcode_set": "(list) - The barcodes this run decodes, one entry per barcode, each naming the barcode, the reference CSV that names its sequences and the regex group it is captured by; an entry may be just the barcode name when a reference CSV is already named beside it. Default blank, which decodes the plate column, the guide and the plate row from column_csv, grna_csv and row_csv, exactly as every run did before this key existed. A set may hold one barcode or ten, and the regex has to name a group for every entry in it, so the run refuses a pattern that names fewer and says which barcode has no group.",
"column_csv": "(path) - CSV mapping column barcodes to well names; it must have 'sequence' and 'name' columns. Reads are matched verbatim against it with no reverse-complementing, so the sequences must be in the same orientation as the reads - run barecodes_reverse_complement on the file if they are not. Unmatched reads get NA for columnID. Default the bundled spacr/resources/data/barcodes_column.csv; barcode QC (sequencing_qc) instead defaults this key to empty, where the reference is optional.",
"row_csv": "(path) - CSV mapping row barcodes to well names; it must have 'sequence' and 'name' columns. Reads are matched verbatim with no reverse-complementing, so the sequences must be in the same orientation as the reads - use barecodes_reverse_complement to flip the file if needed. Unmatched reads get NA for rowID. Default: the bundled spacr/resources/data/barcodes_row.csv.",
"grna_csv": "(path) - CSV mapping gRNA barcode sequences to gRNA names; it must have 'sequence' and 'name' columns. Reads are matched verbatim with no reverse-complementing, so orientation must match the reads (barecodes_reverse_complement flips a file). Rows whose gRNA does not match are written as NA and dropped from the counts. Default: the bundled spacr/resources/data/grna_barcodes.csv.",
"save_h5": "(bool) - Also write every annotated read (consensus sequence plus its parsed row/column/gRNA barcodes and IDs) to annotated_reads.h5. The per-well counts in unique_combinations.csv and qc.csv are written either way, so set it False unless you need read-level data; True produces a very large file and compression can dominate runtime. Default True.",
"comp_type": "(str) - PyTables compression library used when writing annotated_reads.h5, passed to pandas HDFStore as complib: 'zlib', 'lzo', 'bzip2' or 'blosc'. 'blosc' is far faster at similar file size, 'bzip2' is smallest but slowest. Ignored entirely when save_h5 is False. Default 'zlib'.",
"comp_level": "(int) - complevel passed to the HDF5 store, 0-9. 0 disables compression (fastest write, largest file); higher values shrink annotated_reads.h5 at increasing CPU cost, and at the top of the range saving can take longer than the barcode mapping itself. Ignored when save_h5 is False. Default 5.",
"custom_model_path": "(str) - Path to a trained classifier artifact whose model weights initialize a new fine-tuning run. The optimizer and epoch start fresh. Leave empty to initialize from ImageNet or random weights according to init_weights. Default ''.",
"resume_checkpoint": "(str) - Path to a spaCR training artifact to continue exactly: restores model, optimizer, scheduler, epoch, best score and random-generator state. Use custom_model_path instead when only the weights should be reused. Default ''.",
"normalize": "(bool or list) - Control percentile normalization before display, model input, or crop export. Display and activation-map tools use True for a 2nd-to-98th-percentile stretch. Measure and External Masks start at False; Measure accepts False or a two-number [low, high] percentile pair and refuses bare True because it supplies no bounds. It affects display and exported-crop scaling, not measured source intensities. Default True in the display-oriented tools.",
"overlay": "(bool) - In the batch-grid figures, draw the activation map in the 'jet' colormap at 50 percent alpha over the source image. Turn it off and the grid tiles are left empty apart from the predicted-class label, so keep it on whenever plot is enabled. It never affects the per-object activation PNGs saved to disk, which are always the bare map. Default True.",
"normalize_input": "(bool) - Apply the per-channel mean 0.5 and standard deviation 0.5 used during training before generating activation maps. Match this setting to model training; otherwise inputs are out of distribution and the resulting classes and maps are invalid. This is distinct from normalize, which percentile-stretches images for display. Default True.",
"distance_gaussian_sigma": "(int or None) - Sigma in pixels of the Gaussian blur applied to each channel before measuring intensity-weighted centroid distances from cells to nuclei and pathogens. Larger values smooth out speckle so the weighted centroid follows broad signal. None or 0 skips these distance features entirely. Needs a cell mask plus a nucleus or pathogen mask. Default 10.",
"infection_xgb_n_estimators": "(int) - Number of boosting rounds (trees) trained, passed as num_boost_round. More rounds fit the intensity-extreme training set more tightly and push infection probabilities away from 0.5, which shrinks the ambiguous band, but cost runtime and can overfit small wells. Trade off against infection_xgb_learning_rate. Default 200.",
"infection_xgb_max_depth": "(int) - Maximum depth of each boosted tree. Deeper trees capture interactions between morphology and pathogen-intensity features but overfit the quartile-derived training labels; shallower trees generalise better across wells. Typical range 2-8; raise it only when the classifier cannot separate infected from uninfected. Default 3.",
"infection_xgb_learning_rate": "(float) - Shrinkage applied to each boosting round's contribution (XGBoost eta). Lower values require more rounds but can produce smoother, better-calibrated infection probabilities; higher values converge faster but can yield probabilities concentrated near 0 or 1, reducing the utility of the ambiguous range. Typical range 0.01-0.3; tune together with infection_xgb_n_estimators. Default 0.1.",
"infection_xgb_subsample": "(float) - Fraction of training rows drawn at random for each boosting round, between 0 and 1. Below 1 it injects stochasticity that limits overfitting to the small set of intensity-extreme cells used for training; 1.0 uses every training row every round. Lower it if the classifier appears to memorise individual wells. Default 0.8.",
"infection_xgb_colsample_bytree": "(float) - Fraction of feature columns offered to each tree, between 0 and 1. Lowering it stops a couple of dominant pathogen-intensity features from being chosen by every tree, spreading gain across morphology features and reducing overfitting; 1.0 exposes all features to every tree. Default 0.8.",
"infection_xgb_reg_lambda": "(float) - L2 penalty on leaf weights. Larger values shrink leaf outputs, giving a more conservative model whose probabilities sit closer to 0.5 and therefore more cells inside the ambiguous band; 0 removes the penalty entirely. Raise it when the model fits training cells perfectly yet disagrees wildly with mask-based labels. Default 1.0.",
"infection_xgb_random_state": "(int) - Seed for the generator that balances the per-well training set, i.e. which intensity-extreme cells are sampled for each class. It is not handed to XGBoost itself. Change it and re-run to confirm the adjusted infection calls are stable under a different training draw. Default 42.",
"infection_xgb_n_jobs": "(int) - Threads XGBoost uses for training and prediction (its nthread parameter). -1 uses every available core; set a small positive number to leave CPU free for other work or when several plates run at once. It changes runtime, not the training recipe. Default -1.",
"infection_xgb_proba_threshold": "(float) - Predicted probability at or above which a cell is called infected, between 0 and 1. Lowering it makes infection calling more permissive (more cells become infected), raising it more stringent. It is also the centre of the confidence band whose half-width is infection_xgb_margin. Default 0.5.",
"infection_xgb_margin": "(float) - Half-width of the confidence band around infection_xgb_proba_threshold, clamped to 0-0.49. In 'relabel' mode only cells outside the band get their label overridden, the rest keep the mask-based call; in 'remove' mode cells inside the band are spared deletion. Raise it to trust the model less. Default 0.15.",
"infection_xgb_top_features": "(int) - How many features, ranked by XGBoost gain, are retained for the feature-importance panel of the QC figure. This is a display cut applied after training: it never changes the model or the infection calls. Lower it for a readable bar chart, raise it to inspect more features. Default 20.",
"infection_xgb_proba_column": "(str) - Column used by both the track-level ambiguous filter and the QC probability plot. If it is absent, both components discover the classifier output column, normally infection_prob. Before 2026-08-12 this fallback was unreachable, so the track-level filter was not applied under the default configuration. Set this explicitly only to override discovery. Default 'infection_xgb_proba'.",
"infection_xgb_drop_ambiguous": "(bool) - After prediction, discard cells whose probability lies between infection_xgb_ambiguous_low and infection_xgb_ambiguous_high instead of forcing a call on them. True gives cleaner infected vs uninfected motility comparisons at the cost of sample size; False keeps every cell. Only used by the xgboost strategy. Default True.",
"infection_xgb_ambiguous_low": "(float) - Lower edge of the discarded probability band, between 0 and 1. Cells whose probability falls between this and infection_xgb_ambiguous_high are dropped when infection_xgb_drop_ambiguous is True. Raise it toward the threshold to keep more cells, lower it to discard more borderline ones. Swapped automatically if it exceeds the high bound. Default 0.25.",
"infection_xgb_ambiguous_high": "(float) - Upper edge of the discarded probability band, between 0 and 1. Together with infection_xgb_ambiguous_low it defines the interval whose cells are dropped when infection_xgb_drop_ambiguous is True. Lower it toward the threshold to keep more cells, raise it to discard more. Swapped automatically if it falls below the low bound. Default 0.75.",
"infection_xgb_min_cells_per_class": "(int) - Per well, how many intensity-extreme examples each class must reach before that well's training data are balanced by subsampling to the smaller class; wells that have both classes but fewer examples contribute all of theirs, unbalanced. Wells with only one class are skipped entirely. No well is ever excluded for being small, so raising it leaves more wells unbalanced and the training set more skewed - lower it towards 1 to force balancing in every usable well. Default 10.",
"infection_pca_method": "(str) - Records the embedding used ('pca', 'umap' or 't-sne'). The pipeline derives and overwrites this value from infection_intensity_strategy during QC; change infection_intensity_strategy to select the embedding. This output remains empty until QC has run. No default.",
"infection_pca_random_state": "(int) - Seed for KMeans and for the UMAP/t-SNE embeddings in the pca/umap/tsne strategies. Fixing it makes the embedding and the resulting infected/uninfected cluster assignment reproducible; change it to check that the split is not an artifact of one initialisation. Note the max-cells subsample uses its own fixed seed. Default 42.",
"motility_ylim": "(tuple) - Spatial y-axis limits for the origin-centred track panels (infected and uninfected) of the motility figure, in plotted coordinate units - um when pixels_per_um is set, otherwise pixels - not velocity. The whole-field all-tracks axis next to them always autoscales from the data and ignores this setting. Set to None for autoscaling. Default (100, -100), a 200-unit window written high-to-low so the axis draws reversed.",
"motility_xlim": "(tuple) - Spatial x-axis limits for the origin-centred track panels (infected and uninfected) of the motility figure, in plotted coordinate units - um when pixels_per_um is set, otherwise pixels - not time. The whole-field all-tracks axis next to them always autoscales from the data and ignores this setting. Set to None for autoscaling. Default (100, -100), a 200-unit window written high-to-low so the axis draws reversed.",
"seconds_per_frame": "(int) - Interval between consecutive timelapse frames, in seconds. Used with pixels_per_um to convert mean per-frame displacement into um/min; if either is missing, velocities stay in px/frame. It is also printed in the motility plot legend box. A wrong value rescales every reported velocity linearly. Default 60.",
"pixels_per_um": "(float) - Image scale in pixels per micrometre. Track coordinates are divided by it, so plots switch from px to um, and together with seconds_per_frame it converts velocity from px/frame to um/min. Take it from the objective and camera pixel size rather than tuning it - it rescales every reported velocity. Default 1.78.",
"infection_intensity_n_bins": "(int) - Bin count for the pathogen-intensity histogram, clamped to 10-256. The histogram strategy evaluates bins from low to high and uses the first bin whose infected fraction reaches the target as the intensity threshold. More bins provide finer threshold resolution but increase variability in per-bin fractions. This setting also controls the QC-panel histogram. Default 64.",
"db_table_name": "(str) - Table inside <src>/measurements/measurements.db that holds the pre-QC per-frame measurements. It is rewritten with if_exists='replace' on every run, a companion table with the suffix '_well_motility' holds the well summary, and the same name is read back when reuse_existing_measurements is True. Default 'timelapse_object_measurements'.",
"infection_intensity_qc_graphs": "(bool) - Save the infection-intensity histogram PNG and reserve the QC sub-axes (histogram, embedding, or XGBoost probability plus feature importance) inside the combined intensity/motility panel. Set False to skip that plotting work on large runs; the infection relabelling itself is unchanged either way. Default True.",
"infection_intensity_qc_panel_path": "(str) - Path of the QC image embedded in the mask panel. The histogram strategy records its generated PNG; other strategies leave this value empty. The pipeline clears any supplied value before QC, so this is a reported output rather than a control. No default; empty until QC has run.",
"infection_intensity_mode": "(str) - Action applied when the quality-control classification disagrees with the mask-based label. 'relabel' replaces the label and retains the cell; 'remove' excludes cells with discordant mask and intensity evidence. Unknown values fall back to 'relabel'. Default 'relabel'.",
"infection_intensity_strategy": "(str) - How infected vs uninfected is decided once infection_intensity_qc is True: 'xgboost' trains a classifier on intensity extremes, 'histogram' picks one intensity threshold, and 'pca'/'umap'/'tsne' cluster a 2D embedding. Unknown values fall back to histogram, as does xgboost when the package is missing or a class is too small. Default 'xgboost'.",
"infection_intensity_qc": "(bool) - Master switch for infection re-calling. While False the mask-based label (cell contains at least one pathogen) is used unchanged and every other infection_* setting is inert; True runs the method chosen by infection_intensity_strategy. A pathogen_channel must also be set. No default is applied anywhere, so it behaves as False until you set it.",
"straightness_threshold": "(float) - Straightness cut-off, where straightness = net displacement / total path length (0 = returns to start, 1 = perfectly straight). When drop_straight_tracks is True, tracks at or above this value are dropped as drift or tracking artifacts, so lowering it discards more tracks. The count is always logged. Default 0.95.",
"drop_straight_tracks": "(bool) - Apply the straightness threshold. False reports how many tracks exceed straightness_threshold without changing the data; True removes those tracks from the velocity table, per-well summary and plots. Enable it when stage drift or identity swaps produce implausibly straight trajectories. Default False.",
"track_outlier_zscore": "(float) - Outlier sensitivity when smoothing scalar features within a track (area, bbox area, equivalent diameter, perimeter, solidity, mean/max/min intensity). A frame more than this many standard deviations from its own track mean, whose two neighbours are both within half that, is replaced by their average. Lower smooths more; nothing is deleted. Default 3.0.",
"max_displacement": "(float) - Largest plausible centroid movement between consecutive frames, in pixels. A single-frame excursion followed by an immediate return is interpolated from neighbouring positions; other displacements above this value cause the complete track to be excluded. Increase the value for rapidly moving objects or sparsely sampled timelapses and decrease it to remove identity-switch artifacts. Default 50.0.",
"tracked_object": "(str) - Which object's feature block ({object}_* columns) the XGBoost infection classifier trains on: 'cell', 'nucleus' or 'pathogen'; anything else falls back to 'cell'. It does not change what is tracked - track geometry and velocity always come from the cell centroids. Default 'cell'.",
"motility_analysis": "(bool) - Run the automated motility assay once per plate, after all masks are merged: it rebuilds per-object measurements from merged/*.npy, cleans tracks, computes per-track velocity and straightness, applies the infection QC, and writes motility_plots plus a well-level summary table. It only fires when timelapse is also True, and it is what reveals the Motility setting categories. Default False.",
"reuse_existing_measurements": "(bool) - If measurements.db already holds the table named by db_table_name, load it instead of re-extracting regionprops from merged/*.npy. Saves most of the runtime when re-running only the infection QC or the plots, but it also skips track smoothing, so changes to max_displacement or track_outlier_zscore only take effect with this set to False. Default True.",
"infection_pca_umap_search": "(bool) - Fit UMAP once per combination of infection_pca_umap_n_neighbors_grid and infection_pca_umap_min_dist_grid, keeping the run with the highest cluster-centroid distance times ground-truth separation. True costs one UMAP fit per grid point; False does a single fit using infection_pca_umap_n_neighbors and infection_pca_umap_min_dist. Default True.",
"infection_pca_umap_n_neighbors_grid": "(list[int]) - Candidate UMAP n_neighbors values tried when infection_pca_umap_search is True. Small values (around 5) preserve local structure and split fine subpopulations; large values (30 and up) emphasise global structure. Every entry is paired with every value in infection_pca_umap_min_dist_grid, so keep the list short. Default [5, 10, 15, 30].",
"infection_pca_umap_min_dist_grid": "(list[float]) - Candidate UMAP min_dist values tried when infection_pca_umap_search is True, each between 0 and 1. Near 0 packs points tightly and gives crisper clusters for KMeans to split; larger values spread points out and blur the boundary. Paired with every n_neighbors candidate. Default [0.0, 0.05, 0.1, 0.3].",
"infection_pca_pathogen_weight": "(float) - Multiplier applied to the standardised pathogen-channel features before embedding. Above 1 it stretches the embedding along pathogen intensity so KMeans splits infected from uninfected rather than by morphology; 1.0 leaves all features weighted equally. Raise it when the log reports weak cluster separation. Default 2.0.",
"infection_pca_log_intensity": "(bool) - Apply log1p to non-negative features whose names contain 'intensity', 'p75', 'p95', or 'max' before standardization and embedding. This compresses the upper tail and reduces the influence of a small number of high-intensity cells. Consider enabling for pathogen stains with a wide dynamic range. Default False.",
"infection_pca_tsne_search": "(bool) - Fit t-SNE once per combination of infection_pca_tsne_perplexity_grid and infection_pca_tsne_learning_rate_grid, keeping the run that scores highest on centroid distance times ground-truth separation. False does a single fit at infection_pca_tsne_perplexity with learning_rate 'auto'. Every extra grid point costs a full t-SNE fit. Default True.",
"infection_pca_tsne_perplexity_grid": "(list[float]) - Candidate t-SNE perplexity values tried when infection_pca_tsne_search is True - roughly how many neighbours each point balances. Candidates at or above (n_cells-1)/3 are discarded, and if none survive the code falls back to min(30, that cap). Small values fragment clusters, large ones merge them. Default [15.0, 30.0, 45.0].",
"infection_pca_tsne_learning_rate_grid": "(list[float]) - Candidate t-SNE learning rates tried when infection_pca_tsne_search is True. Too low leaves a dense ball with points crowded together; too high scatters the map into a diffuse cloud. Either way the infected/uninfected split blurs. Paired with every perplexity candidate, so keep both lists short. Default [200.0, 500.0].",
"infection_pca_umap_n_neighbors": "(int) - Fixed UMAP n_neighbors used when infection_pca_umap_search is False; it defines the size of the local neighbourhood UMAP attempts to preserve. Low values (5-10) emphasize local detail and can divide one population into multiple clusters; values of 30 or greater emphasize global structure. Ignored during grid search. Default 15.",
"infection_pca_umap_min_dist": "(float) - Fixed UMAP min_dist used when infection_pca_umap_search is False, between 0 and 1: the minimum spacing allowed between embedded points. Near 0 gives tight, well-separated clumps that KMeans splits cleanly; larger values spread points evenly and blur the infected/uninfected boundary. Ignored during grid search. Default 0.1.",
"infection_pca_tsne_perplexity": "(float) - Fixed t-SNE perplexity used when infection_pca_tsne_search is False, automatically capped at max(5, (n_cells-1)/3). Lower values emphasise local structure and can break one population into several clumps; higher values emphasise global structure and merge them. The learning rate is left at 'auto'. Default 30.0.",
"infection_pca_min_silhouette": "(float) - Silhouette value below which the log prints a 'weak cluster structure' warning with tuning hints. It does not reject or re-run the clustering - the cluster-derived labels are applied regardless - so treat it purely as an alert level. Silhouette runs from -1 to 1. Default 0.05.",
"infection_pca_min_gt_separation": "(float) - Alert level for the ground-truth separation score - the absolute difference, between the two clusters, in the fraction of intensity-extreme cells that are infected (0-1). Dropping below it only prints a warning; the cluster labels are still applied. Raise it to be told sooner that the embedding is not separating infection. Default 0.2.",
"infection_pca_max_cells": "(int) - Maximum number of cells included in the embedding. When more cells are available, a random subsample of this size is drawn with a fixed seed of 0, independently of infection_pca_random_state. Decrease the value to reduce UMAP or t-SNE runtime and memory use; increase it to improve representation of rare subpopulations. Applied after removal of non-finite rows. Default 50000.",
'number_of_organelles': "(int) - How many organelle slots this run has, "
"from 0 to 26. Each slot is an independent object with its own "
"channel, its own type preset and its own copy of every detection "
"setting, matching the pattern organelle(?:[b-z]|[a-z]{2})?_.*; "
"raising the number generates another slot's settings and lowering "
"it hides the slots above the new number without deleting them. A "
"hidden slot keeps its values, is still written to the settings "
"file, and comes back exactly as it was when the number is raised "
"again, so a smaller number can be tried without losing work. "
"Default 0.",
'organelle_channel': "(int) - Zero-indexed raw acquisition channel segmented into organelle masks by whichever organelle_method is chosen (otsu, adaptive, log, dog, ridge, hysteresis, cellpose, unet). Setting it to an integer adds an organelle mask plane to merged/ and unlocks the Organelle setting categories in the GUI; None skips organelle segmentation entirely. Default None.",
'organelle_type': "(str) - Organelle morphology used to populate recommended detection settings; explicitly configured values are not overwritten. Options are 'punctate', 'vesicular', 'spherical', 'filamentous', 'tubular', 'reticular', 'cisternal', 'toroidal' and 'crescent'. Morphology alone does not determine the detector: 'vesicular' and 'spherical' also use organelle_diameter because a 200 nm vesicle appears punctate whereas a 2 µm vacuole appears annular. Default 'custom', which applies no recommendations.",
'organelle_morphology': "(str) - Shape family of the target organelle; selects the segmentation pipeline and restricts valid organelle_method values. 'spots' denotes punctate structures such as vesicles and lipid droplets; 'network' denotes filamentous structures such as mitochondria and endoplasmic-reticulum tubules; 'irregular' denotes solid, irregular structures such as Golgi and lysosomes; and 'ring' denotes hollow structures such as endosomes and autophagosomes. An unsupported morphology-method pair raises before image loading. Default 'spots'.",
'organelle_method': "(str) - Segmentation backend, validated against organelle_morphology: 'otsu' (one global threshold), 'adaptive' (local threshold), 'log'/'dog' (blob detection), 'ridge' (tubeness filter, network only), 'hysteresis' (dual threshold, network only), 'cellpose' (pretrained model), 'unet' (your own model, network only). Classical methods run on CPU across n_jobs workers; cellpose and unet run on the GPU. Default 'otsu'.",
'organelle_diameter': "(float) - Deprecated. Expected organelle diameter in pixels. The Cellpose-SAM path used for organelles calls model.eval with diameter=None, and no classical method sizes its kernels from it, so changing this value has no effect on organelle masks. Bound object size with organelle_min_area / organelle_max_area instead. Default 30.",
"organelle_model_name": "(str) - Cellpose model used when organelle_method='cellpose'. Cellpose 4 provides only 'cpsam'; the pre-SAM names are accepted and mapped to it. Change this only to point at a custom CPSAM-architecture checkpoint. Of the three parameters that used to distinguish models only diameter still acts (eval rescales by 30/diameter); model_type and diam_mean are logged 'not used in v4.0.1+' and dropped. Default 'cpsam'.",
'organelle_remove_border': "(bool) - Delete organelle labels touching any image edge in both Mask runs and Live Preview. Enabled when either this setting or organelle_remove_border_objects is True; the Live Preview toggle updates both names together. Removes partially imaged objects from area and intensity statistics, at the cost of losing valid objects at the field boundary. Default False.",
'organelle_log_min_sigma': "(float) - Smallest Gaussian scale searched by LoG blob detection, in pixels; the detected blob radius is about sigma times sqrt(2), so sigma 1 finds roughly 1.4 px radius puncta. Raise it to ignore single-pixel noise, lower it to catch the smallest spots. Default 1; must stay below organelle_log_max_sigma.",
'organelle_log_max_sigma': "(float) - Largest Gaussian scale searched by LoG blob detection, in pixels; blob radius is about sigma times sqrt(2), so 10 caps detection near a 14 px radius. Raise it to catch large puncta, at a runtime cost since the filter is evaluated once per scale. Default 10; must exceed organelle_log_min_sigma.",
'organelle_log_num_sigma': "(int) - How many Gaussian scales are evaluated between organelle_log_min_sigma and organelle_log_max_sigma. More scales resolve a wider spread of spot sizes, but the filter runs once per scale so runtime grows linearly. Default 10; drop to 3-5 when spot size is uniform and you need speed.",
'organelle_log_threshold': "(float) - Minimum LoG/DoG response a local maximum must reach to count as a blob, measured after the image is percentile-normalised to 0-1, so it behaves like a contrast fraction. Decrease it to detect fainter puncta at the cost of additional noise; increase it to retain only brighter puncta. Default 0.01. The 'dog' method also reads this key.",
'organelle_tophat_radius': "(int) - Radius in pixels of the disk used for white top-hat filtering before Otsu or adaptive spot thresholding; it removes structures broader than the disk, reducing haze and background. Set it slightly above the largest expected spot: smaller values suppress spots, whereas larger values retain more background. Default 5. Ignored by the LoG and DoG methods.",
'organelle_watershed_spots': "(bool) - Split touching spots instead of labelling each connected blob once. Under otsu/adaptive it runs a distance-transform watershed with seeds at least 5 px apart; under log/dog it grows a watershed from each blob centre instead of stamping a disk whose radius comes from that blob's own sigma (round(sigma*sqrt(2)), minimum 1 px). Turn it off when single spots are being fragmented. Default True.",
'organelle_ridge_sigmas': "(list of float) - Scales in pixels at which the vesselness filter detects tubular structures; each value should approximate the half-width of a filament, and responses are combined across scales. Add larger values to detect thick bundles and retain smaller values for fine tubules. Default [1, 2, 3]; runtime increases approximately in proportion to list length.",
'organelle_ridge_filter': "(str) - Which vesselness filter enhances filaments before thresholding: 'frangi' (classic, crisp on well-separated tubules), 'sato' (more tolerant of varying thickness), 'meijering' (tuned for thin neurite-like fibres). All run with black_ridges=False, i.e. bright filaments on a dark background. Default 'frangi'; try 'sato' when frangi drops faint filaments.",
'organelle_skeletonize': "(bool) - Reduce each thresholded network to a one-pixel-wide skeleton (dilated by 1 px so it stays connected) and label that instead of the filled filaments. Measured areas then track network length rather than filament thickness. Enable for topology and length analysis, disable to measure filament mass. Default False.",
'organelle_network_threshold': "(str) - How the ridge-filter response is binarised: 'otsu' takes one global cut-off from the response histogram, 'adaptive' uses a local threshold (organelle_adaptive_block_size / _offset) and keeps faint filaments in dim regions at the cost of extra background. Only read by organelle_method='ridge'; anything unrecognised falls back to otsu without warning. Default 'otsu'.",
'organelle_adaptive_block_size': "(int) - Side length in pixels of the local neighbourhood used to compute the adaptive threshold; must be odd. Small blocks track fine illumination changes but can carve holes out of large organelles; large blocks behave more like a global threshold. A few times the object diameter is a sensible starting point. Default 51.",
'organelle_adaptive_offset': "(float) - Subtracted from each local mean to form the adaptive threshold, so a pixel is foreground when it exceeds local_mean minus this value. Increasing the offset lowers the threshold and produces more foreground; use a small or negative value for stricter segmentation. The value uses raw image-intensity units, so an offset tuned for 16-bit data can oversegment ridge and ring modes, which threshold a 0-1 response. Default 5.",
'organelle_morph_radius': "(int) - Radius in pixels of the disk used for morphological cleanup. In irregular mode it also sets the pre-smoothing sigma (radius/2) and drives a closing then an opening, bridging gaps and erasing protrusions thinner than the disk; network modes use half this radius for closing only. Raise it to smooth ragged outlines, lower it to preserve fine detail. Default 3.",
'organelle_fill_holes': "(int) - Fill interior holes up to this area in square pixels after thresholding, preventing a darker centre from creating a ring-shaped segmentation artifact. This is applied only in irregular mode. Increase it when large organelles are incorrectly hollow; use a low value or 0 when a hollow centre is biologically expected. Default 64.",
'organelle_cellprob_threshold': "(float) - Cellpose cellprob_threshold. Pixels whose predicted probability of belonging to an object fall below it are excluded, so raising it shrinks masks and drops faint organelles, while lowering it grows masks and recovers dim ones along with more false positives. Useful range roughly -6 to 6. Default 0.0.",
'organelle_flow_threshold': "(float) - Cellpose flow_threshold: maximum error allowed between a candidate mask's flows and the network prediction. Lower values discard more irregular masks; higher values retain more irregular objects. Increase it when valid non-round organelles are being discarded. Default 0.4.",
'organelle_resample': "(bool) - Deprecated. Passed to Cellpose as resample: when True the flows are recomputed at full resolution instead of on the downsampled grid, giving smoother and slightly more accurate outlines for a little extra time. Still forwarded to model.eval on the organelle path. Default True; retain the default unless reduced runtime is required.",
'organelle_mask_dim': "(int) - Position along the last axis of each merged/*.npy array where the organelle label mask sits. Masks follow the image channels in the order cell, nucleus, pathogen, organelle, so with four channels and all three other masks present it is 7. Leave it unset/None and organelles are not measured at all. No default is applied.",
'organelle_chann_dim': "(int or None) - Legacy merge field retained for older settings files. Current Mask uses organelle_channel to decide whether to append the mask plane, and Measure uses organelle_mask_dim to locate that plane; changing this field has no effect. Default None.",
'organelle_rolling_ball': "(bool) - Apply rolling-ball background estimation with organelle_rolling_ball_radius, subtract the estimated background, and clip negative values to zero before segmentation. This corrects uneven illumination and haze so that a single global threshold can be applied across the field of view, at an additional computational cost per image. Default False.",
'organelle_rolling_ball_radius': "(int) - Radius in pixels of the rolling-ball background estimator. It must exceed the diameter of the largest expected organelle to avoid subtracting the objects themselves; an excessively large value may not follow the illumination gradient. An initial value of several times the expected object diameter is appropriate. Default 50; runtime increases steeply with radius.",
'organelle_clahe': "(bool) - Rescale each image to 0-1 on its 0.5/99.5 percentiles, then run contrast-limited adaptive histogram equalisation before segmentation. Pulls dim organelles in dark corners up to the same working contrast as bright ones, at the cost of amplifying background noise and destroying absolute intensity comparability between fields. Default False.",
'organelle_clahe_clip_limit': "(float) - Contrast ceiling for CLAHE, range 0-1: each tile's histogram is clipped at this height before equalisation, so higher values permit stronger local stretching and more noise amplification. 0.01 is gentle, 0.03-0.05 is aggressive. Only read when organelle_clahe is True. Default 0.01.",
'organelle_mask_within_cells': "(bool) - Zero every pixel outside the cell mask before segmenting, so organelles can only be found inside cells and extracellular debris cannot generate objects. Needs cell_mask_stack/ to already exist alongside the organelle source; if it is missing spacr prints a warning and carries on unmasked rather than failing. Default False.",
'organelle_dog_sigma_low': "(float) - Smallest Gaussian scale searched by Difference-of-Gaussians blob detection, in pixels; it sets the lower bound on detectable spot size (radius about sigma times sqrt(2)). Raise it to suppress fine noise, lower it to catch the smallest puncta. Default 1.0. The detection cutoff itself comes from organelle_log_threshold, not from a dog-specific key.",
'organelle_dog_sigma_high': "(float) - Largest Gaussian scale searched by Difference-of-Gaussians blob detection, in pixels. Scales are stepped up from the low sigma by a factor of 1.6 until this bound, so widening the gap costs more passes but covers a wider range of spot sizes. Raise it to catch larger spots. Default 3.0; must exceed organelle_dog_sigma_low.",
'organelle_hysteresis_low': "(float) - Weak threshold for hysteresis segmentation: pixels above it are kept only where they connect to a seed above organelle_hysteresis_high. Values below 1.0 are read as a fraction and converted to that percentile of the smoothed image (0.2 = 20th percentile); 1.0 or above is an absolute intensity. Lower it to trace filaments further into their dim tails. Default 0.2.",
'organelle_hysteresis_high': "(float) - Strong threshold that seeds hysteresis segmentation - only components containing a pixel above it survive at all, then they grow outward down to organelle_hysteresis_low. Values below 1.0 are read as a percentile of the smoothed image (0.6 = 60th percentile); 1.0 or above is absolute. Raise it to keep only confidently bright filaments. Default 0.6.",
'organelle_unet_model_path': "(str or None) - Path to a serialised PyTorch model used when organelle_method='unet'. It must be a torch.load-able whole module, not a state_dict, and take z-scored (B,1,H,W) input returning (B,1,H,W) logits; extra output channels are silently ignored except the first. A missing or invalid path raises before segmentation starts. Default None.",
'organelle_unet_threshold': "(float) - Probability cut-off applied to the U-Net's sigmoid output, range 0-1. Lower it to grow the predicted network and recover faint branches at the cost of false positives; raise it to keep only confident pixels, which tends to break weak connections. Objects below organelle_min_area are still removed afterwards. Default 0.5.",
'organelle_ring_sigma_inner': "(float) - Low sigma of the Difference-of-Gaussians band-pass that highlights ring walls, in pixels; set it near the wall thickness so the wall survives the high-pass. Too small and pixel noise is retained, too large and the wall blurs into the lumen and the ring stops being detected as hollow. Default 1.0; must be below organelle_ring_sigma_outer.",
'organelle_ring_sigma_outer': "(float) - High sigma of the ring Difference-of-Gaussians band-pass, in pixels; it sets the coarse scale that gets subtracted, so keep it around the ring's outer radius. Widen the gap from organelle_ring_sigma_inner to enhance larger rings, narrow it for tight vesicles. Default 3.0; must exceed organelle_ring_sigma_inner.",
'organelle_ring_min_prominence': "(float) - Shape gate for ring mode: for each filled object spacr computes abs(mean wall intensity minus mean lumen intensity) divided by the object's mean intensity, and deletes anything below this value. Raise it to keep only clearly hollow objects, lower it to also accept partly filled ones. 0 disables the gate. Default 0.1.",
'organelle_ring_fill_method': "(str) - How detected ring walls become solid objects: 'flood' fills every background component that does not touch the image border - accurate, but leaks through any gap in the wall - while 'convex' takes the convex hull of each wall component, which tolerates broken rings but overshoots concave shapes. Default 'flood'; switch to 'convex' when rings come out unfilled.",
'summarize_organelles_by': "(str, list or None) - Parent compartments to roll every enabled organelle slot into. Accepts 'cell', 'nucleus', 'pathogen' and 'cytoplasm'; each writes one <parent>_organelle_summary row per parent with a separate organelle_summary_<slot>_* column family. Raw per-organelle tables are always written when their mask dim is enabled. Default 'cell'; None disables only these rollups.",
'cell_perimeter_fraction': "(float) - For each touching pair of cell labels, the shared boundary length divided by the smaller object's perimeter; pairs at or above this fraction are merged into one cell. Low values such as 0.1 merge aggressively and can fuse true neighbours, high values only rejoin pieces of the same cell. 0 disables perimeter merging. Default 0.",
'nucleus_perimeter_fraction': "(float) - Merge two touching nucleus labels when their shared boundary covers at least this fraction of the smaller object's perimeter. Low non-zero values merge aggressively (0.1 joins barely-touching nuclei); high values only fuse objects sharing most of an edge. Range 0-1; 0 (default) disables perimeter merging. Use it when one nucleus is split into fragments.",
'pathogen_perimeter_fraction': "(float) - Fraction, from 0 to 1, of the smaller label's perimeter that two touching pathogen objects must share before they are merged. The default of 0 disables perimeter-based merging. Values near 0.1 merge most touching objects, whereas values from 0.5 to 0.8 merge only objects with a long shared boundary. Use this setting to join vacuoles that Cellpose divided into multiple labels.",
'organelle_perimeter_fraction': "(float) - Merge two touching organelle labels when their shared boundary is at least this fraction of the smaller object's perimeter. Range 0-1; increase it toward 1 to merge only nearly fully fused pairs, or decrease it to merge labels with shorter shared boundaries. Applied before area and mean-intensity filtering in both Mask runs and Live Preview. Default 0 (disabled).",
'organelle_min_area': "(int) - Post-segmentation area floor in square pixels; smaller objects are deleted and the mask relabelled. Raise it to clear noise specks left by thresholding, lower it to keep faint puncta. One filter for both the live preview and the batch run. Default 10 in Mask; Measure and External Masks start at 0 because they consume existing labels rather than segmenting new ones.",
'organelle_max_area': "(int or None) - Post-segmentation area ceiling in square pixels; larger objects are deleted rather than split. Use it to reject fused clumps, saturated debris and background merged by Otsu into one component. Values below the largest valid organelle remove biological objects without warning. One filter for both the live preview and the batch run. Default None in Mask, meaning no limit; 0 in Measure and External Masks, which also disables it.",
'organelle_min_intensity': "(float) - Delete organelle objects whose mean pixel intensity in organelle_channel is below this value, measured in the original image's raw units. Equality is retained. Raise it to reject dim objects. Applied after segmentation in both Mask runs and Live Preview. 0 disables this bound. Default 0.",
'organelle_max_intensity': "(float) - Delete organelle objects whose mean pixel intensity in organelle_channel is above this value, measured in the original image's raw units. Equality is retained. Lower a positive bound to reject more bright objects. Applied after segmentation in both Mask runs and Live Preview. 0 disables this bound. Default 0.",
'cell_remove_border_objects': "(bool) - Delete every cell label touching any of the four image edges before measurement. Removes partial cells whose area and total intensity are truncated and would bias per-cell statistics, at the cost of losing objects - a large cost in fields where cells are big relative to the field. Default False.",
'nucleus_remove_border_objects': "(bool) - After segmentation, delete every nucleus label touching any of the four image edges, then renumber the rest. Enable it when measuring nucleus area or total intensity, since clipped nuclei bias those downward; leave it off for counts or positions, as it discards real objects at every field boundary. Default False.",
'pathogen_remove_border_objects': "(bool) - Delete any pathogen label touching the first or last row or column of the image. Enable it so partially imaged parasites do not enter area and intensity statistics with truncated values; leave it off when parasites are sparse and losing edge objects costs too much data. Default False.",
'organelle_remove_border_objects': "(bool) - Delete organelle labels touching any image edge in both Mask runs and Live Preview. Enabled when either this setting or organelle_remove_border is True; the Live Preview toggle updates both names together. Removes partially imaged objects from area and intensity statistics, at the cost of losing valid objects at the field boundary. Default False.",
"annotation_column": "(str) - Integer column in the png_list table that stores manual class labels. The Annotate app adds it with ALTER TABLE if it is absent and writes labels to it. This column provides the reference labels when dataset_mode is 'annotation' and is the fallback when annotation_columns is unset. Supplying it while dataset_mode is unset also selects annotation mode for compatibility with older settings files. Default None.",
'cmap': "(str) - Matplotlib colormap applied to single-channel image previews and plate heatmaps. Perceptually uniform maps ('viridis', 'inferno', 'magma') preserve the relative visibility of intensity differences; 'gray' resembles the raw single-channel microscope image. Any registered matplotlib name is accepted, with an '_r' suffix to reverse it. Default 'inferno' for image plots and 'viridis' for plate heatmaps.",
'nontargeting_control_grnas': "(list) - Non-targeting control gRNA identifiers. Their coefficients set the volcano effect-size cutoff: abs(median(control coefficients)) + threshold_multiplier × spread, with threshold_method selecting the spread estimator. A wider control distribution raises the cutoff; None disables it. Default ['000000'] names the non-cutting control gene, which spaCR resolves to all associated guides in the loaded library. Individual guide identifiers also work, with or without the organism prefix.",
'count_data': "(str or list) - CSV(s) of per-well gRNA read counts from the sequencing step (unique_combinations.csv); each must contain grna, count, rowID and columnID columns or the run raises ValueError. These are the regression's independent variable. Pass one path per plate, position-aligned with plates_count; results are written under the first file's folder. Default 'list of paths', a placeholder that must be replaced; the barcode QC module defaults this key to 'path to unique_combinations.csv'.",
'cov_type': "(str) - Heteroscedasticity-robust covariance estimator passed to likelihood fits: 'HC0', 'HC1', 'HC2', 'HC3', or None for classical non-robust errors. It changes standard errors and P-values, not coefficients. Use 'HC3' when residual variance increases with well cell count. Penalized, robust and quantile fits do not support this estimator and raise an error rather than reporting ordinary errors under a robust label. Default None.",
"resume": '(bool) - Continue an interrupted run from its last validated boundary. Mask revalidates existing mask and merged arrays; Measure accepts only fields complete in every owned table and clears partial rows before retrying; and Format Converter reopens each checkpointed TIFF. These validations reduce the risk of reusing partial output and require additional reads during resumption. Default False.',
"resume_search": "(bool) - Continue the compatible Image UMAP hyperparameter checkpoint stored under the project results folder. Completed trial scores and embedding arrays are loaded without refitting. If an adaptive 2x2 round was interrupted, only its missing corners are evaluated before the direction is chosen. The feature data, labels, search space, criterion, seed, increments and stopping threshold must match. Default False.",
"checkpoint_path": "(str or None) - Optional explicit path for an atomic resume checkpoint. Format Converter defaults to .spacr_conversion.checkpoint.json in its destination; Image UMAP defaults to results/.spacr_checkpoints/umap_search.json under the project. Keep checkpoints with their outputs. Default None.",
"umap_stability_repeats": "(int) - Number of independently seeded embeddings fitted for each configuration in multi-objective UMAP search. Stability is the mean fraction of k nearest neighbours shared between every pair of repeats, so it is unaffected by rotation or reflection. Runtime scales linearly; minimum 2, default 3. API: spacr.hyperparam.embedding_stability.",
"umap_neighborhood_weight": "(float) - Relative multi-objective weight for neighborhood preservation, defined as the geometric mean of trustworthiness and continuity so both invented and lost neighbours are penalized. Weights are normalized to sum to one. Default 0.4. API: spacr.hyperparam.umap_objective_scores.",
"umap_stability_weight": "(float) - Relative multi-objective weight for repeat-to-repeat nearest-neighbour stability. Weights are normalized to sum to one. Default 0.3. API: spacr.hyperparam.umap_objective_scores.",
"umap_cluster_structure_weight": "(float) - Relative multi-objective weight for positive silhouette structure. spaCR uses supplied labels when available; otherwise it reports the best reproducible K-means silhouette across 2-8 clusters. This can reveal candidate structure but cannot prove biological meaning. Default 0.3. API: spacr.hyperparam.umap_objective_scores.",
'background_correction': "(str) - Per-object local background subtracted from the outside-stain statistic before thresholding. 'auto' uses the median of the five-pixel ring outside the parasite mask, removing a per-field offset without a flat-field image; 'none' performs no subtraction. A brightly stained attached parasite may produce an antibody halo that extends into the reference ring, in which case subtraction can reduce the signal required for classification. Default 'none'.",
'bimodality_cutoff': '(float) - Minimum bimodality coefficient required for an unflagged field- or well-level efficiency estimate. Values below the threshold are retained but flagged because the intensity distribution provides insufficient evidence for two populations. Increasing the threshold requires stronger bimodality. Default 0.5555555555555556.',
'change_plate': "(bool) - Relabel each source directory as plate1, plate2, ... instead of trusting the plate ID stored in its database. Use it when several plates were written under the same name, which would otherwise let two plates' fields pool into one threshold and one well. Default False.",
'compartment': "(str) - Prefix used by per-object measurement columns, so 'pathogen' selects pathogen_area and pathogen_channel_1_percentile_95. It must match the object type contained in the table; otherwise the run stops and reports the unresolved area and intensity columns. Default 'pathogen'.",
'control_quantile': "(float) - Quantile of the control wells' outside-stain distribution used as the threshold. A value of 0.99 classifies approximately one percent of genuinely unstained parasites as attached. Lowering it toward 0.95 reduces false invaded classifications while increasing false attached classifications; raising it has the opposite effect. Default 0.99.",
'stain_baseline_wells': "(list or None) - These wells set the empirical negative distribution for the pre-permeabilisation stain and are excluded from efficiency calculations. Set a column ('c12'), row ('r1'), well ('r1_c12'), or complete plate key whose parasites received no stain; None uses the automatic per-field method. Default None.",
'analysis_excluded_wells': "(list or None) - Wells dropped before anything is fitted: the sequencing sweep removes them from the count table before it looks for the fraction threshold, and the regression never sees them. It must name the same wells as filter_value or the threshold is fitted on wells the fit has already dropped. Regression initializes it from filter_value plus any declared control blocks (the shipped default is ['c1', 'c2', 'c3']). Set a column ('c12'), row ('r1'), well ('r1_c12'), or complete plate key.",
'extracellular_class': "(str) - Classification policy for parasites that overlap no host cell. 'attached' assigns them to the attached class independent of stain intensity because they cannot be intracellular; 'classify' uses the stain signal when host-cell segmentation is uncertain; 'exclude' removes them before summary calculations. n_no_host_cell reports their count under every policy. Default 'attached'.",
'group_column': "(str) - Column whose values become the experimental conditions compared against each other; 'condition' is the combined host-cell / pathogen / treatment label built from the plate-metadata maps. Point it at 'pathogen' or 'treatment' to compare on one factor alone. Rows with no value here are dropped before anything is counted. Default 'condition'.",
'inflation_warn': '(float) - Additional invasion efficiency, in proportion units, that increasing the threshold by threshold_sensitivity may add to a well before the well is flagged. Only the upward change is monitored because decreasing the threshold can only reclassify invaded parasites as attached and cannot create a positive invasion result. A value of 0.05 flags a well whose efficiency would increase by more than five percentage points. Default 0.05.',
'intensity_statistic': "(str) - Per-object statistic of the pre-permeabilisation channel used for thresholding. Because the stain is localized to the parasite surface, the object mean divides rim signal by the full area and can classify a larger parasite as dimmer than a smaller parasite with equivalent surface staining. A percentile of rim-pixel intensity reduces this size-dependent bias. Default 'mean'. Invasion starts at 'auto': it chooses periphery_95 when present, otherwise percentile_95, and uses mean only as a warned last resort.",
'level': "(str) - Result level. For regression, 'both' writes results_grna.csv and results_gene.csv and corrects each family separately; 'grna' reports guide effects, and 'gene' pools guides by gene. Nonparametric inference also honours this choice. Mixed models disable it because they estimate gene effects with guides nested inside genes. For proportion plots, the same key selects 'object', 'well', or 'plate' aggregation. Default 'both' for regression and 'object' for proportions.",
'max_parasite_area': '(float or None) - Largest object area in pixels kept as a parasite. Anything bigger is several parasites merged by the mask, whose rim statistic mixes them and whose single classification then stands for all of them. None keeps everything. Default None.',
'min_control_objects': "(int) - Minimum number of objects required from a plate's control wells before their quantile is used as a threshold. Below this value, the plate uses the automatic per-field method and records the fallback rather than estimating a 99th percentile from an insufficient sample. Default 10.",
'min_objects_for_bimodality': '(int) - Objects required before the bimodality coefficient is computed at all; below it the coefficient is left NaN and the field or well is flagged. The statistic exceeds its cutoff on genuinely unimodal data about 45% of the time at ten objects and 15% at twenty, so computing it there would silence the check exactly where the classification is least trustworthy. Default 30, where that false-pass rate is 5%.',
'min_objects_for_threshold': "(int) - Minimum number of objects required to derive a field-specific threshold. Below this count, the field uses its well threshold and then its plate threshold; automatic_source records the level used. Increasing the value improves statistical stability but reduces adaptation to local illumination variation. Default 10.",
'min_parasite_area': '(int or float) - Smallest object area in pixels retained as a parasite. Smaller objects are treated as debris because their outside-stain statistic is estimated from too few pixels for stable thresholding. Increase it when pathogen masks are over-segmented into small fragments. Default 0, which applies no area filter.',
'min_parasites_per_well': "(int) - Minimum number of scored parasites required for an unflagged well-level efficiency estimate. At n=50 and p=0.5, the normal-approximation 95% interval has an approximate half-width of 13.9 percentage points. Estimates below the threshold remain in the output with a flag and their n_total value. Default 50.",
'min_total_intensity': '(float or None) - Minimum mean intensity in the post-permeabilisation channel for an object to count as a parasite at all. That antibody stains every parasite, so an object dark in it is debris inside the pathogen mask rather than a dim parasite, and it would otherwise contribute a background-level outside signal and be scored invaded. None applies no filter. Default None.',
'outside_channel': "(int) - Zero-indexed channel of the pre-permeabilisation antibody, which labels parasites remaining outside the host cell. Classification thresholds this channel, so an incorrect index changes the assay readout to the signal measured in another channel without raising an error. This is an image-channel index, not measure.py's <object>_channel_<n>_outside_* columns, which quantify the ring outside an object's mask. Default 1.",
'outside_threshold': '(float or None) - Fixed threshold for the outside-stain statistic, applied to every field and overriding both the automatic method and control wells. Use only after independent calibration: a value above the true threshold misclassifies attached parasites as invaded and inflates invasion efficiency. Control wells, when supplied, remain the reference against which quality control evaluates the fixed value. None derives the threshold per field. Default None.',
'outside_threshold_method': "(str) - Method used to derive the outside-stain threshold from each field when neither a fixed value nor control wells are supplied: 'otsu', 'triangle', 'li', 'yen', or 'mean'. These methods identify a partition but do not test whether the distribution is bimodal; that evaluation is performed separately. 'triangle' is appropriate for a strongly skewed distribution with a small stained minority, whereas 'otsu' is appropriate for a more balanced distribution. Default 'otsu'.",
'parasite_table': "(str) - Table in measurements/measurements.db holding one row per segmented parasite. It is read directly rather than through the usual merge, because that merge collapses pathogen rows onto their host cell and would sum several parasites' stain intensities into a single row. Change it only if measure_crop wrote the parasite objects under a non-standard name. Default 'pathogen'.",
'vacuole_key': "(str) - Rule used to group individually segmented parasites into vacuoles. 'auto' prefers an explicit vacuole-ID column, otherwise spatially clusters centroids, then falls back to host cell or one parasite per vacuole with a warning. Set 'spatial', 'cell_id', 'object', or an explicit column name to make that biological assumption reproducible. Default 'auto'.",
'replication_method': '(str) - Readout for spacr.submodules.analyze_replication. direct_count counts individually segmented parasites per assigned vacuole. size_proxy delegates to the existing area-derived endodyogeny analysis, which aggregates pathogen area per host cell and assumes one vacuole per host for a vacuole interpretation; it does not count parasites or measure volume. deep_learning_coming_soon is reserved for the forthcoming whole-vacuole classification model and cannot run yet. Default direct_count.',
'vacuole_link_distance': "(float or None) - Maximum centroid-to-centroid distance in pixels for spatially linking parasites into the same vacuole. None derives the distance from median parasite diameter times vacuole_link_factor; set a calibrated value when magnification or segmentation scale varies between plates. Too large merges separate vacuoles and too small splits one rosette. Default None.",
'vacuole_link_factor': "(float) - Multiplier applied to the median segmented-parasite diameter when vacuole_link_distance is derived automatically. Increasing it joins wider rosettes but also raises the risk of merging nearby vacuoles; decreasing it does the reverse. It is ignored when an explicit link distance or vacuole-ID column is used. Default 1.5.",
'parasite_count_column': "(str or None) - Optional column that already stores the number of parasites represented by each segmented row. When set, the assay sums that column per vacuole instead of counting rows, which is required if one row can represent several parasites. None treats every retained row as one parasite. Default None.",
'max_parasites_per_vacuole': "(int) - Largest power-of-two parasite count given its own replication bucket. Counts above it remain visible in a '>N' bucket and non-powers stay in the separate QC bucket; they are never clipped or rounded. Use a power of two large enough for the experiment's duration. Default 16.",
'require_host_cell': "(bool) - Drop parasite rows with no valid host-cell link before constructing vacuoles. This prevents extracellular debris and attached parasites from entering a replication readout, but it will also remove real infected cells when cell segmentation or parent assignment failed. The number removed is reported. Default True.",
'non_power_of_two_warn': "(float) - Fraction of a well's vacuoles allowed in the non-power-of-two bucket before the well is flagged as unreliable. Three-, five-, or seven-parasite rosettes usually indicate segmentation or vacuole-linking errors, so lowering the threshold makes QC stricter without deleting any observations. Default 0.2.",
'qc_plot_max_panels': '(int) - Largest number of wells drawn in the threshold-diagnostic figure, taken in sorted well order. It exists so a 384-well plate does not produce a 384-panel figure; the CSVs always carry every well regardless. Default 12.',
'seed_wells_from_cells': '(bool) - Read the cell table as well, so a well holding host cells but no parasites appears in the results with a zero denominator instead of vanishing from the plate entirely. Switch it off only when the database has no cell table. Default True.',
'threshold_agreement_tolerance': "(float) - Relative distance a threshold may sit from its reference before the field and well are flagged; the reference is the control-derived cut when controls exist, otherwise the field's own automatic cut. 0.5 means a factor of two. Lower it to catch smaller drifts between a fixed threshold and what the data would have chosen. Default 0.5.",
'threshold_sensitivity': "(float) - Fractional amount the threshold is moved up and down to produce the invasion_efficiency_low_threshold and _high_threshold bracket, which shows how much of a well's answer is the threshold rather than the biology. Widening it widens the bracket and makes the inflation flag more eager. Default 0.25.",
'total_channel': '(int or None) - Zero-indexed channel of the post-permeabilisation antibody that stains every parasite. Nothing is classified from it; it only supplies the intensity that min_total_intensity filters on, so an incorrect value costs nothing until that filter is switched on. Default 0.',
'attribution_baseline': "(str) - Replacement used when deletion/insertion curves remove a pixel: 'blur', 'zero', or 'noise'. Replacing pixels with zero can create an out-of-distribution edge, causing part of the score change to reflect the replacement artifact rather than removed information. 'blur' generally produces the smallest distribution shift and is the default. Comparing multiple baselines quantifies the sensitivity of the AUC to this choice. Default 'blur'.",
'attribution_steps': '(int) - Points along the deletion and insertion curves used to score a map. At each step the highest-ranked remaining pixels are removed or added and the model is re-evaluated. This sets the resolution of the area under the curve used to assess whether the map identifies image features contributing to the prediction. More steps produce a smoother AUC with linearly more forward passes. Default 12.',
'ig_baseline': "(str) - Reference image used by integrated gradients: 'zero' is black, 'blur' is a blurred copy of the input image, and 'noise' is random. The baseline defines the attribution reference and therefore changes the result. On dark-field images, a zero baseline attributes broadly to bright object signal; a blurred baseline preserves low-frequency content and emphasizes contributions from image detail. Default 'zero'.",
'ig_steps': "(int) - Interpolation steps between the baseline and input image for integrated gradients. The completeness approximation improves with step count; too few steps increase the error without raising an exception. Validate the completeness error when reducing the default of 50. Computational cost is linear in this number. Default 50.",
'object_type': "(str) - Mask used to define an object when the pointing game scores an attribution map: 'cell', 'nucleus', 'pathogen' or 'cytoplasm'. The metric checks only whether the map's maximum-valued pixel lies inside that mask. It has low computational cost but does not evaluate the rest of the map, so a method can score 1.0 while assigning spurious attribution elsewhere. Default 'cell'.",
'occlusion_stride': '(int) - How far the occlusion patch moves between evaluations. Equal to occlusion_window it tiles without overlap and is fastest; half of it doubles the passes and halves the blockiness. A stride larger than the window leaves unmeasured gaps that appear as an artificial grid in the map. Default 4.',
'occlusion_window': "(int) - Side length in pixels of the patch moved across the image during occlusion analysis. Larger windows reduce runtime but spatial resolution and can miss features smaller than the window; smaller windows resolve finer structure with quadratically more forward passes. Occlusion provides a gradient-independent comparison for gradient-based attribution methods. Default 8.",
'counterfactuals': '(bool) - Also train a small class-conditional generator on the crops, guided by the loaded classifier, and morph held-out crops toward the other class in steps. Writes each crop\'s classifier score along its sequence, the flip rate, how far the edit moved the crop, a class-mean-shift baseline and a figure to counterfactuals/ next to the maps. The same classifier guides and scores the edits, so read the flip rate with the edit size. Default False.',
'counterfactual_crops': '(int) - How many crops, taken in dataset order, train and test the counterfactual generator; a quarter is held out for scoring. More crops give a steadier estimate and a slower run. Ignored unless counterfactuals is on. Default 256.',
'counterfactual_epochs': '(int) - Training passes of the counterfactual generator over its crops. Ignored unless counterfactuals is on. Default 30.',
'counterfactual_condition': "(str) - What the counterfactuals morph between: 'class' uses the classifier's classes; 'plate', 'well', 'row' or 'column' uses that condition, read from each crop's file name, and pushes every edit toward the classifier's mean class probabilities for the target condition, so a control well's cells are drawn the way the classifier sees a treated well's. Needs at least two conditions among the crops. Ignored unless counterfactuals is on. Default 'class'.",
'counterfactual_target': "(str) - Which class or condition every counterfactual morphs toward: a class index such as 1 when counterfactual_condition is 'class', otherwise a condition name as read from the crop file names, such as a well. Cells already in the target are left out. Empty morphs each cell to the next class or condition in sorted order. Ignored unless counterfactuals is on. Default ''.",
'counterfactual_generator': "(str) - The model that draws the counterfactuals. 'autoencoder' is fast and trained to flip the classifier. 'diffusion' is trained only to draw crops of each class and redraws a partly noised crop under the target class, so a flip is not something it was optimised for; it is slower and needs a GPU, thousands of crops and hundreds of epochs; with a few hundred crops it barely edits. Diffusion weights are saved as counterfactual_diffusion.pt next to the tables. Ignored unless counterfactuals is on. Default 'autoencoder'.",
'sanity_check': "(bool) - Randomize the model's weights layer by layer, recompute attribution and report the similarity between maps. A method that produces nearly the same map for a randomized model is responding to image structure rather than the trained decision function. On a small CNN, the CAM family, including spaCR's default Grad-CAM, fails this test while saliency and integrated gradients pass. The resulting similarity is reported for the selected model rather than inferred from benchmark behavior. This costs one additional attribution per randomized layer. Default True.",
'smoothgrad_samples': "(int) - Number of noise-perturbed image copies averaged into one attribution map. Using 8-50 samples reduces local gradient variability and improves between-image comparability. A value of 0, the default, evaluates the method once and minimizes computation during method selection. Applies to every method, including the CAM family, where maps are averaged explicitly rather than through Captum.",
'smoothgrad_sigma': "(float) - Standard deviation of the noise added by SmoothGrad, expressed as a fraction of the image intensity range. Values that are too small produce nearly identical samples and little averaging effect; values that are too large move samples outside the training distribution, causing the average to characterize responses to noise rather than the experimental images. Values of 0.1-0.2 are typical. Ignored when smoothgrad_samples is 0. Default 0.15.",
'strict_errors': "(bool or None) - Error-handling policy for recoverable steps. Off records failures in the run ledger and final summary while continuing with successful items. On raises immediately for setup or configuration errors such as unreadable paths, missing columns or inaccessible databases, preventing partial batch results from invalid inputs. Per-item failures such as one corrupt image remain recoverable under either policy. None defers to $SPACR_STRICT_ERRORS. Default None.",
'max_failure_rate': "(float or None) - Fraction of failed items above which the run aborts. For example, 0.2 aborts after more than 20% of items fail. The failure ledger is written to the artifact before the abort. None disables rate-based abortion; failures remain counted and reported, and incomplete artifacts are marked partial. Default None.",
'queue_by_uncertainty': "(bool) - Order the Annotate grid by increasing classifier confidence rather than database order. Samples near the decision boundary generally provide more information for active learning than samples assigned 99% confidence. Requires model scores in png_list; when none are available, the grid uses page order and reports the fallback. Previously annotated crops are excluded. Default False.",
'queue_measure': "(str) - Method used to score uncertainty for the annotation queue. 'entropy' incorporates all classes and is the preferred default for three or more classes; 'least_confidence' ranks samples by the highest class score; 'margin' ranks them by the difference between the two highest scores. With exactly two classes, margin and least_confidence produce identical rankings, including ties, and diverge only with three or more classes. These values are uncertainty scores rather than calibrated probabilities. Default 'entropy'.",
'queue_diversity': "(str) - Metadata level across which the annotation queue is distributed. Ranking solely by uncertainty can concentrate the highest-ranked crops in one or two wells, repeatedly sampling the same source of ambiguity. Distribution across wells or plates increases experimental coverage at a modest cost in per-item uncertainty. Default 'well'.",
'queue_limit': "(int) - Maximum number of crops in the annotation queue. 0 queues the complete unlabelled pool. When the limit is smaller than the number of wells, queue_diversity selects approximately one crop from each represented well rather than the globally highest-uncertainty crops. Default 0.",
'class_balance': "(str) - Correction for imbalance among training classes. 'none' preserves sampling and reports class counts, their ratio and a recommendation. 'weighted_sampler' samples classes approximately equally using 1/n. 'sqrt_weighted_sampler' uses 1/sqrt(n), reducing the risk that repeated sampling of a very small class causes memorization. 'weighted_loss' preserves sampling and weights the loss. Use one of the latter three when the reported ratio exceeds approximately 3:1. Default 'none'.",
'cross_validation': "(bool) - Score the classifier with 5-fold stratified cross-validation instead of a single train/test split, so every control object receives an out-of-fold prediction and an optimal probability threshold is picked per fold. Gives a far more stable accuracy estimate on small control sets, at roughly 5x the training time. Default True.",
'cross_validation_folds': "(int) - Number of k-fold splits used to train the vision classifier instead of one val_split holdout. 0 (the default) or 1 uses one random split; 2 or more trains a separate model per fold, evaluates each model on its held-out fold, and reports the mean, fold-to-fold standard deviation, and range. These statistics quantify sensitivity to the data partition. Runtime is approximately proportional to k. Distinct from 'cross_validation', which controls the regression pipeline.",
'cross_validation_enabled': "(bool) - Enable k-fold validation for Classify. If cross_validation_folds is 0 or 1, enabling this uses 5 folds. Use cv_group_by='plate' to hold out whole plates, or 'well'/'field' for within-plate validation without leaking related crops between training and validation. Default False.",
'cv_group_by': "(str) - Train/test independence: 'cell', 'field', 'well' (default), or 'plate'. Cell can place sibling crops from one well on both sides; field narrows but does not close that leak; well matches the usual experimental assignment unit; plate holds out a complete batch. Whole groups make the requested fraction approximate, so runs report held-out groups and cells. Legacy 'none'/'off' alias 'cell'. Crop identities come from spaCR's plate_well_field_object.png names; unverifiable grouped designs are refused rather than silently randomized.",
'classifier_evaluation': "(bool) - Build the Classifier Evaluation workbench bundle from out-of-fold predictions after Classify (CV): sample-level predictions, confusion matrices, reliability curves, calibrated probabilities, per-plate metrics, leakage reports and a manifest. It requires cross_validation_folds >= 2; a single train/validation split cannot produce unbiased out-of-fold diagnostics. Default True. API: spacr.classifier_evaluation.evaluate_predictions.",
'nested_cv_inner_folds': "(int) - Number of inner grouped folds used inside every outer CV fold. 0 (default) keeps the faster ordinary grouped CV; 2 or more trains one inner model per fold, uses inner validation for early stopping/model selection, ensembles those models, and evaluates only once on the untouched outer fold. Runtime is approximately outer_folds x inner_folds training runs, but the outer score is not reused for tuning. API: spacr.classifier_evaluation.nested_group_folds.",
'evaluation_calibration': "(str) - Probability calibration written to the evaluation bundle. 'temperature' cross-fits one scalar temperature per held-out fold using all other out-of-fold predictions, so a sample never fits its own calibrator; 'none' retains raw softmax probabilities. Calibration changes reported probabilities, not the saved model weights. Default 'temperature'. API: spacr.classifier_evaluation.cross_calibrate_probabilities.",
'evaluation_bins': "(int) - Number of equal-width probability bins in reliability curves and expected calibration error. Values around 10 balance resolution against noise; use fewer bins for small validation sets and more only when every class has many hundreds of out-of-fold samples. Minimum 2, default 10. API: spacr.classifier_evaluation.calibration_table.",
'evaluation_fail_on_leakage': "(bool) - Stop Classify (CV) before fitting a fold when the same object, augmentation family, or protected cv_group_by identity appears in both train and validation. False records the problem and continues, which is useful only for diagnosing a legacy dataset because its performance estimate remains invalid. Default True. API: spacr.classifier_evaluation.audit_split_leakage.",
'leakage_audit_train_test': "(bool) - Audit the permanent train/ and test/ boundary before any classifier fit. Checks plate/well/field/object lineage, exported augmentation families and (when enabled) byte-identical renamed copies. Default True. API: spacr.classifier_evaluation.audit_dataset_splits.",
'leakage_hash_content': "(bool) - SHA-256 hash classifier images during leakage audits so an identical crop copied or renamed across train/test or CV boundaries is still detected. Reads files in 1 MiB chunks and never decodes pixels. Default True. API: spacr.classifier_evaluation.audit_cv_folds.",
'leakage_require_identity': "(bool) - Treat filenames that do not encode the protected cv_group_by identity, and files that cannot be hashed, as a failed audit rather than an advisory warning. Default True because independence cannot be claimed when lineage is unknown. API: spacr.classifier_evaluation.audit_split_leakage.",
'early_stopping_patience': "(int) - Stop training after this many consecutive epochs in which validation accuracy fails to beat the best value so far; the best checkpoint is still kept. 0 (default) disables it and always runs the full 'epochs' budget. Set 10-20 on long runs to cut wasted epochs once the model plateaus.",
'tensorboard': "(bool) - Write PyTorch loss, accuracy, macro-F1, and learning-rate events to dst/tensorboard while the vision model trains. Run tensorboard --logdir <path> with that directory to open an interactive dashboard and compare runs. The in-app loss and accuracy monitor is controlled separately by plot. Default True.",
'filter_column': "(str) - Metadata column used to drop control wells before regression: every row whose value appears in filter_value is removed from both the score data and the read counts. Use 'columnID' (default) when controls sit in plate columns, 'rowID' when they sit in rows. In annotate_filter_vision it instead names the score column thresholded by upper_threshold/lower_threshold.",
'filter_min_max': "(list) - Display-only size filter for plot_merged: one [min_area, max_area] pair in pixels per mask dimension, in the order cell, nucleus, pathogen, e.g. [[500,50000],[100,5000],[10,2000]]. Objects outside a pair are erased from that mask before the overlay is drawn. None (default) keeps every object.",
'filter_value': "(list) - Values of filter_column whose rows are removed - not kept - before regression, normally the control columns; default ['c1','c2','c3']. Dropping them stops control wells from dominating the gene and gRNA fits. Only list values take effect: a bare string is silently ignored and nothing is filtered.",
'focal_alpha': "(float) - Class-balancing weight for focal loss (read only when loss_type resolves to focal). In the single-logit binary path it scales positives by alpha and negatives by 1-alpha, so raise it toward 1 to emphasise a rare positive class; with two or more output classes a plain float scales the whole loss uniformly. Default None (no alpha weighting).",
'focal_gamma': "(float) - Focusing exponent in the focal-loss weight (1 - p_t)^gamma, applied only when loss_type is focal. 0 reduces it to cross-entropy; increasing it, typically within 1-5, reduces the contribution of confidently classified crops and increases the relative contribution of difficult samples. Default 2.0. Increase when class imbalance causes training to be dominated by easy examples.",
'generate_training_dataset': "(bool) - Rebuild the train/ and test/ PNG folders from the object crops before training, using the dataset_mode rules (annotation_column labels, metadata rules or measurement rules) and splitting off test_split of the images. Turn it off to reuse an existing split; it is only consulted when train or test is True, and a failed build aborts training. Default True.",
'label_smoothing': "(float) - Epsilon passed to cross-entropy when loss_type is label_smoothing: each target retains 1 - eps of its probability mass and distributes the remainder across the other classes. Increase it, typically within 0.05-0.2, when predicted probabilities are overconfident or annotations are noisy; 0 disables smoothing. Ignored by other loss types. Default 0.1.",
'log_x': "(bool) - Use a log10 x-axis; for line graphs, the x column is log10-transformed rather than only changing the axis scale. Enable when x spans several orders of magnitude, such as gRNA fraction thresholds or count distributions, to prevent low values from being visually compressed. Values at or below zero cannot be displayed. Default False.",
'log_y': "(bool) - Put the y-axis on a log10 scale; for line graphs the y column is log10-transformed instead of the axis being rescaled. Enable it when the measured values span orders of magnitude or a few large wells compress everything else toward the baseline. Values at or below zero cannot be shown. Default False.",
'logit_adjust_tau': "(float) - Strength of the Menon-et-al. logit adjustment: tau * log(class prior) is added to the logits during training, pulling decisions toward rare classes. Only used when loss_type resolves to logit_adjust_ce, which 'auto' picks when the smallest class is under 10% of the data. Higher tau corrects harder; 0 disables. Default 1.0.",
'loss_type': "(str) - Loss used to train the classifier. For a head with two or more classes: 'focal_loss' (down-weights easy examples), 'cross_entropy', 'label_smoothing' (epsilon 0.1), 'ce_weighted' (inverse-frequency class weights), 'logit_adjust_ce' and 'asl'. 'binary_cross_entropy_with_logits' is valid only for a single-logit head and raises otherwise. Use a weighted or focal loss for imbalanced classes. Default 'focal_loss'. Merged Classifier starts at 'auto', which resolves to cross_entropy for a multi-class head and binary_cross_entropy_with_logits for a single-logit head.",
'metadata_files': "(list) - Gene-annotation CSVs, each with a 'Gene ID' column, that are joined onto the regression results by gene, writing an extra results CSV per file. These are gene tables, not plate/well metadata. When toxo is True the order matters: index 0 is read as the ME49 transcription table and index 1 as the GT1 phenotype table. Default [].",
'paired_data': "(list of dicts) - Regression input table: each row explicitly pairs one score CSV with one count CSV. Plate identity comes from both files when they agree, from the partner when only one declares plateID, or from the row order when neither does. A conflict is refused. Legacy score_data/count_data lists are migrated positionally and logged. Default [].",
'min_observations_per_hit': "(int) - Observation count a significant hit must strictly exceed to appear in results_significant_filtered.csv: gRNA hits need n_grna > min_observations_per_hit, gene hits need n_gene > min_observations_per_hit. The unfiltered hit list is still written alongside it. Raise it to drop hits resting on one or two wells. Default 0, which filters nothing.",
'normalization_percentiles': "(list) - Two-element [low, high] percentile pair used to stretch each channel's non-zero pixels to the full display range in plot_merged; applied only when normalize is True. Narrowing the pair (e.g. [5, 95]) boosts contrast but saturates bright objects; widening it flattens the image. Default [2, 98].",
'nr_imgs': "(int) - Number of object crops in each representative-image grid. The sampler selects this many per condition, or all available crops when fewer exist. Increase it for a more representative but larger and slower figure; decrease it for a faster preliminary view. Must be a positive integer; plotting helpers default to 16.",
'nucleus_chann_dim': "(int) - Recruitment analysis only (analyze_recruitment): the image-channel index paired with the nucleus mask when drawing outline overlays, and the switch that enables nucleus_size_range / nucleus_intensity_range filtering. Set it to None to skip nucleus filtering. It plays no part in segmentation - use nucleus_channel for that. Default 0.",
'nucleus_intensity_range': "(list) - Two-element [min, max] bound on mean nucleus-channel intensity used by the recruitment analysis to drop rows from the measurement table - it filters measured objects, not masks or normalization. Rows are kept only if min < mean intensity < max (raw units), and each bound is ignored unless it is an int. Default [0, 100000].",
'nucleus_size_range': "(list) - Two-element [min, max] bound in pixels^2 on nucleus_area, used by the recruitment analysis to drop rows from the measurement table; masks are left untouched. Rows are kept only if min < area < max, and each bound is ignored unless it is an int. Default [0, 100000]; None widens it to [0, 1e100].",
'offset_start': "(int) - Bases to shift from the start of the target_sequence match to the start of the extracted window; negative values move upstream to capture a barcode preceding the anchor. The start is clamped at position 0, so an over-negative value silently shifts the reading frame and the regex stops matching. Default -8.",
'optimizer_type': "(str) - PyTorch optimizer used by deep_spacr.train_model: 'adamw', 'adam', 'adamax', 'sgd', 'rmsprop', 'nadam', 'radam', 'adagrad', 'adadelta' or 'asgd'. AdamW is the robust fine-tuning default; SGD can generalise better but usually needs more epochs. amsgrad applies only to Adam/AdamW. API: spacr.deep_spacr.train_model(optimizer_type=...). Default 'adamw'.",
'schedule': "(str) - Learning-rate scheduler used by spacr.deep_spacr.train_model: 'cosine', 'cosine_warm_restarts', 'reduce_lr_on_plateau', 'step_lr', 'exponential', 'linear', or 'none'. Plateau reacts to validation loss; cosine and linear use the epoch budget; warm restarts periodically raise the rate to escape a narrow minimum. API: train_model(schedule=...). Default 'cosine'.",
'outlier_detection': "(bool) - After building the regression table, drop gRNAs whose well count falls outside 1.5x the 5th-95th percentile spread, then recompute the per-gRNA tables. This removes gRNAs present in implausibly few or many wells that would otherwise dominate coefficients; disable it if your library is deliberately uneven. Default True.",
'columnID': '(str) - Plate-map field identifying the plate column, the 1-24 axis of a well name. It is used to group wells and define the heatmap x-axis. Selecting a different field produces an incorrectly defined axis without raising an error. Default None.',
'control_sgrnas': '(list) - gRNA names treated as controls when computing the mixed-condition fraction; every fraction is measured relative to these controls. An incorrect or incomplete list shifts every fraction on the plate in the same direction without raising an exception. Verify the list against the screening library. Default None.',
'csv': '(str, path) - Sequencing-derived CSV of gRNA counts per well, used to compute the mixed-condition fraction the scores are compared against. Distinct from csv_name, which names the per-plate score files; this one is a single file for the whole comparison. Default None.',
'csv_name': '(str) - Filename searched for within each entry of folders. All plates must use the same name. A plate whose file has a different name is omitted from the heatmap. Default None.',
'cv_csv': '(str, path) - CSV of cross-validated scores, added to the heatmap as its own row so a held-out score can be read directly beside the in-sample ones. Leave it None to plot the score rows alone. Default None.',
'data_column': '(str) - Column in each per-plate score CSV that provides the heatmap values. It must exist in every file named by csv_name; a plate missing this column is omitted. Selecting an incorrect column produces an unintended visualization without necessarily raising an error. Default None.',
'data_column_cv': '(str) - Column in cv_csv holding the cross-validated value. Named separately from data_column because a cross-validation file almost always labels its column differently from the score files it came from. Default None.',
'feature_importance': '(bool) - Fit a random forest against the score column and plot its impurity-based importances. Fast and always available, but biased toward high-cardinality and correlated features, so read the result as a shortlist rather than a ranking. Turning it off skips that plot and its two grouped-by-compartment and grouped-by-channel companions. Default True.',
'filter_1': '(list or None) - Two-element [column, minimum] filter applied before counting. Rows whose column value is not strictly greater than minimum are excluded. Use it to remove objects that are too small or dim to score so they do not affect the reported percentage. This strict > filter has one lower bound and no upper bound. Default None, which disables pre-filtering.',
'folders': '(list) - One folder per plate holding the classification CSVs to combine into a single heatmap; each is searched for csv_name. The order given is the order the plates are laid out in. Default None.',
'fraction_grna': '(str) - The single gRNA whose fraction is kept and plotted, selected out of the fraction table once it has been computed. Everything else in that table is discarded, so this picks which gRNA the comparison is about. Default None.',
'include_all': "(bool) - When grouping feature importances by compartment and channel, also emit an 'all' row containing features that belong to no single compartment or channel. When disabled, those features are omitted and grouped bars may not sum to the total importance. Default False.",
'permutation_importance': '(bool) - Re-score the fitted forest after shuffling each feature in turn for ten repeats, then rank features by the resulting score decrease. This is slower than feature_importance but avoids its impurity-based bias. Enable it when the feature ranking will support a substantive conclusion. Default False.',
'scores': "(str, path) - CSV of per-object model scores to interpret, joined to measurements on plateID, rowID, columnID, fieldID and object_label. This must be the output of the classification run under interpretation. A CSV from a different model or plate may still satisfy the join keys and produce attribution results for mismatched predictions without raising an error. Default None.",
'shap': '(bool) - Compute SHAP values, which attribute each individual prediction to each feature instead of ranking features overall. It is the only one of the three that can explain a single object, and by far the slowest - see shap_sample before enabling it on a full plate. Default False.',
'shap_sample': "(bool) - Run SHAP on a subsample rather than every object. Runtime and memory use increase with row count. Subsampling reduces computational cost but omits local explanations for excluded objects. Disable when attribution is required for every object. Default True.",
'value_col': "(str) - Measurement column compared with threshold to classify each object as positive. Objects strictly above the threshold are annotated 'above'; the remainder are annotated 'below', and the reported percentage per well is the fraction above. Select the column representing the phenotype, such as a recruitment ratio or mean intensity. Results from different value_col settings are not directly comparable. Default None; must be supplied.",
'outline_color': "(str) - Three-letter code choosing the RGB colours for cell, nucleus and pathogen outlines, in that order: 'rgb', 'bgr', 'gbr' or 'rbg'. The default 'gbr' draws cells green, nuclei blue and pathogens red. An unrecognised string falls back to 'rbg' without warning, so a typo changes your figure colours rather than raising. Change it when an outline clashes with a channel. Default 'gbr'.",
'crop_dtype': "(str) - Data type used for saved crop files. 'original' preserves the pipeline output, including uint16 data from a 16-bit camera. 'uint8' converts using the same high-byte rule as the PNG path. 'uint16' casts an 8-bit crop without rescaling because multiplication by 257 would change measured intensities without adding information. This setting controls storage size and compatibility with other software. Training precision is unchanged because ToTensor converts the input to floating point in [0,1]. Default 'original'.",
'input_mean': "(list or None) - Per-channel means for input_statistics='custom' or 'dataset', on the 0-1 scale ToTensor produces. One value is broadcast to every channel, which is what a single-stain dataset wants. Compute the dataset ones with spacr.normalization.dataset_statistics rather than guessing. Ignored unless input_statistics asks for them. Default None.",
'input_std': "(list or None) - Per-channel standard deviations paired with input_mean. Zero values are replaced with 1.0 to prevent division by zero and non-finite training losses; a zero standard deviation indicates a constant channel. Default None.",
'input_statistics': "(str) - Mean and standard deviation used by the loader when normalize_input is enabled. 'symmetric' uses spaCR's historical 0.5/0.5 convention (Inception and TF-Slim), mapping [0,1] to [-1,1]. 'imagenet' and 'clip' use the distinct statistics associated with those pretrained backbones. 'dataset' estimates per-channel statistics from the current data and may be appropriate for fluorescence images whose intensity distribution differs from photographs. 'custom' uses input_mean and input_std. Default 'symmetric'.",
'outline_palette': "(str) - Colour palette used for object outlines in overlay figures. 'default' assigns red to cells, blue to nuclei, green to pathogens, and yellow to organelles. 'colourblind' uses vermillion, sky blue, bluish green, and yellow from the Okabe-Ito palette. Under deuteranope simulation, the minimum pairwise separation is 27 of 255 for 'default' and 142 for 'colourblind'. Default 'default' preserves the appearance of existing figures.",
'outline_thickness': "(int) - Width in pixels of the mask outlines on the merged overlay; the contour is drawn at this thickness and then dilated by a square of the same size, so the visible line is roughly twice the value. Raise it for large fields where a 1-2 px outline disappears. Default 3.",
'overlay_chans': "(list) - Exactly three channel indices from the stack, mapped in order onto the red, green and blue planes of the merged overlay. Default [1, 2, 3] puts channel 1 in red, 2 in green and 3 in blue; reorder or repeat indices to change which stain reads as which colour. Indices past the stack's channel count are ignored.",
'pathogen_chann_dim': "(int) - Recruitment analysis only (analyze_recruitment): the image-channel index paired with the pathogen mask when drawing outline overlays, and the switch that enables pathogen_size_range / pathogen_intensity_range filtering. Set it to None to skip pathogen filtering. It plays no part in segmentation - use pathogen_channel for that. Default 2.",
'pathogen_intensity_range': "(list) - Two-element [min, max] mean-intensity filter applied to the pathogen table in analyze_recruitment; pathogens whose mean intensity in the paired mask channel falls outside the open interval are dropped before recruitment ratios are computed. Bounds must be ints - floats are silently ignored. Default [0, 100000]. Use it to exclude dead or saturated parasites.",
'pathogen_loc': "(list of lists) - Well locations of each pathogen condition, one inner list per name in pathogens, read by annotate_filter_vision when labelling vision-model score CSVs. Every entry must be a row or column ID string such as 'c1' or 'r3'; ranges are not expanded and unmatched entries leave those wells NaN. Set it alongside pathogens, or leave both None. Default None.",
'pathogens': "(list) - Names of the pathogen conditions scored by annotate_filter_vision, e.g. ['wt','mutant']. Element i is written into the pathogen column for every well in pathogen_loc[i] and folded into the combined condition label. Must match pathogen_loc element for element; if pathogen_loc is None, only the first name is applied to every row. Default None.",
'path_string': "(str) - Substring that must occur in a crop path for that crop to enter the dataset, for example 'cell_png' or 'nucleus_png'. The legacy name png_type remains accepted, although the setting performs only path filtering. Default 'cell_png'.",
'crop_source': "(str) - Select where image crops come from. Viewers use 'png' (LOAD IMAGES) for exported crops in data/ or 'merged' (STREAM IMAGES) to cut from merged/*.npy using the measurements database. These viewer modes correspond to training's 'load_images' and 'stream_images' sources. spaCR reports any fallback, and controls that do not apply to the selected source are disabled. Default 'png' in viewers and 'load_images' in training. Image UMAP starts at 'auto', preferring exported PNGs when available and otherwise streaming from merged arrays.",
'object_array': "(str) - On-demand crops: which object the crops are cut around - 'cell', 'nucleus', 'pathogen', 'cytoplasm' or 'organelle'. Its mask plane in merged/*.npy is what defines each object's extent. Default 'cell'.",
'coordinate_columns': "(list or None) - Database columns used to locate an object for on-demand cropping. Two columns may give a centroid row and column; one column may identify a labeled object whose mask supplies its extent. Coordinate-backed routes support only bounding-box crops. None uses the configured mask plane instead. Default None. Merged Classifier derives one identifier from object_array, initially ['cell_id'], so the single-column form is valid rather than an incomplete coordinate pair.",
'crop_shape': "(str) - 'bounding_box' cuts the smallest rectangle containing the object; 'object' masks everything outside it away. Database-sourced crops can only be bounding boxes. Default 'bounding_box'.",
'red_channel': "(int) - Streamed crops only: which plane of the merged array is drawn in the picture's red channel. Any plane may be named, not only the first three, so planes 1, 2 and 4 of a five-plane array is a mapping rather than a slice. A crop already written to disk was coloured when it was written, so this is greyed out when crops are loaded. Default 2, spaCR's shipped mapping.",
'green_channel': "(int) - Streamed crops only: which plane of the merged array is drawn in the picture's green channel. Name the plane holding the stain you want to read as green; it is a choice of source plane, not a position in the array. A crop already written to disk was coloured when it was written, so this is greyed out when crops are loaded. Default 1.",
'blue_channel': "(int) - Streamed crops only: which plane of the merged array is drawn in the picture's blue channel. Set it together with the red and green choices, because the three together decide which stain reads as which colour. A crop already written to disk was coloured when it was written, so this is greyed out when crops are loaded. Default 0.",
'png_type': "(str) - Object crop type selected from the png_list table when building the training dataset; a row is retained only if its PNG path contains this substring. Use 'cell_png', 'nucleus_png', 'pathogen_png', 'cytoplasm_png' or 'organelle_png' to train on whole cells, nuclei, parasites, cytoplasm or organelles. It must match a crop_mode saved by measure_crop. Default 'cell_png'.",
'prune_features': "(bool) - Before training, keep only the top_features columns with the highest ANOVA F-score against the control labels (sklearn SelectKBest with f_classif). Speeds up fitting and can curb overfitting on small control sets, but discards features the model might have used and scores each feature in isolation, ignoring interactions. Default False.",
'reg_alpha': "(float) - L1 penalty on leaf weights for the gradient-boosted classifier (XGBoost and LightGBM; ignored by the other model_type_ml choices). Raising it drives more leaf weights to exactly zero, shrinking the model and its effective feature set - raise it when training accuracy far exceeds test accuracy. Any value >= 0. Default 0.1.",
'reg_lambda': "(float) - L2 penalty on leaf weights for the gradient-boosted classifier (XGBoost, LightGBM, and CatBoost's l2_leaf_reg). Raising it shrinks all weights smoothly rather than zeroing them, damping the influence of any single feature and curbing overfitting, at the risk of underfitting if pushed too far. Any value >= 0. Default 1.0.",
'score_data': "(str or list) - CSV(s) of per-object or per-well phenotype scores, typically from generate_ml_scores. Each must contain dependent_variable. Pass one path per plate, position-aligned with plates_score. The score filename does not name the output folder: runs go under src/results, named for the inference or regression kind and then suffixed _1, _2, and so on. When src is unset, results is created beside the first count_data file. Default 'list of paths'.",
'single_direction': "(str) - Which mate to scan when mode is 'single': 'R1' or 'R2'. The chosen file is read as-is with no reverse-complementing, so selecting 'R2' means target_sequence and regex must be written in R2 orientation or nothing will match. Ignored when mode is 'paired'. Default 'R1'.",
'target_unique_count': "(int) - Desired mean number of distinct gRNAs per well. spaCR evaluates 1000 read-fraction thresholds, selects the threshold whose per-well mean unique-gRNA count has the smallest absolute difference from this value, and discards every gRNA call below that fraction. Decrease it for a stricter well assignment or increase it to retain more gRNAs per well. Default 5.",
'threshold_method': "(str) - Select the spread estimator for the control-based effect-size cutoff: 'std', legacy 'var' (squared units), 'mad', 'iqr', 'percentile' (the 95th percentile of absolute coefficients), or 'range'. 'none' disables the effect-size cutoff. Historical aliases such as 'standard_deveation', 'variance', and 'quantile' are accepted. Used only when controls are set. Default 'std'.",
'threshold_multiplier': "(float) - Set how many control-distribution spreads are required for a hit. The cutoff is abs(median(control coefficients)) + threshold_multiplier × spread, using threshold_method for the spread. Larger values demand a larger effect; threshold_method='none' disables the cutoff. Used only when controls are set. Default 3.",
'use_checkpoint': "(bool) - Run the backbone's forward pass through torch.utils.checkpoint: intermediate activations are discarded and recomputed during the backward pass, trading extra compute for a large drop in activation memory. Enable when a bigger batch_size or image_size gives CUDA out-of-memory; disable for the fastest epochs when VRAM is not the constraint. Default True.",
'x_lim': "(list) - Two-element [min, max] limits on the coefficient (x) axis of the Toxoplasma volcano plot produced by the regression pipeline when annotation_source is the bundled Toxoplasma annotation. Narrow it to zoom in on hits clustered near zero, widen it to keep large-effect genes on the plot. Leaving it None falls back to [-0.5, 0.5], not auto-scaling. Default None."
}
_clone_organelle_registry(tooltips, tooltip=True)
for _role in ORGANELLE_SLOT_ROLES[1:]:
tooltips.setdefault(
_background_switch_key(_role),
tooltips['remove_background_organelle']
.replace('organelle_', f'{_role}_')
.replace('the organelle channel',
f'the organelle {organelle_number(_role)} channel'))
def _name_the_family_in_every_estimator_tooltip():
"""Append "Read by: <families>" to each per-family estimator setting.
Read ownership from ``REGRESSION_SETTINGS_USED``, the same registry used
to enable each control. This keeps the concise footer consistent with the
visible state and automatically covers newly registered families.
"""
from .regression_spec import REGRESSION_SETTINGS_USED
families = {}
for family, keys in REGRESSION_SETTINGS_USED.items():
for key in keys:
families.setdefault(key, []).append(family)
for key, owners in families.items():
text = tooltips.get(key)
if not text or 'Read by regression_type' in text:
continue
if 'Read only by regression_type' in text:
continue
listed = ', '.join(f"'{one}'" for one in sorted(owners))
tooltips[key] = (
f"{text.rstrip()} Read by regression_type {listed}.")
_name_the_family_in_every_estimator_tooltip()
timelapse_settings = ['fps', 'timelapse_mode', 'trackastra_model', 'trackastra_linking', 'ultrack_max_distance', 'ultrack_division_weight', 'ultrack_contour_sigma', 'ultrack_n_workers', 'timeflows_model', 'timelapse_displacement', 'timelapse_memory', 'timelapse_frame_limits', 'timelapse_remove_transient', 'timelapse_objects', 'timelapse_lineage', 'timelapse_lineage_color_by', 'timelapse_lineage_max_distance', 'timelapse_lineage_min_division_h', 'timelapse_events', 'timelapse_events_annotations', 'timelapse_events_model', 'timelapse_events_window', 'timelapse_events_threshold', 'timelapse_events_conditions', 'timelapse_events_encoder', 'timelapse_events_video_checkpoint', 'timelapse_events_video_channels', 'timelapse_events_video_device']
motility_settings = ['motility_analysis','tracked_object', 'infection_intensity_strategy', 'seconds_per_frame', 'pixels_per_um', 'motility_ylim', 'motility_xlim', 'infection_intensity_qc_scope']
motility_advanced_settings = ['reuse_existing_measurements', 'infection_xgb_min_cells_per_class', 'infection_xgb_n_estimators', 'infection_xgb_max_depth', 'infection_xgb_learning_rate', 'infection_xgb_subsample', 'infection_xgb_colsample_bytree',
'infection_xgb_reg_lambda', 'infection_xgb_random_state', 'infection_xgb_n_jobs', 'infection_xgb_proba_threshold', 'infection_xgb_margin', 'infection_xgb_top_features', 'infection_xgb_proba_column',
'infection_xgb_drop_ambiguous', 'infection_xgb_ambiguous_low','infection_xgb_ambiguous_high','infection_pca_method', 'infection_pca_random_state', 'infection_intensity_n_bins', 'db_table_name',
'infection_intensity_qc_graphs', 'infection_intensity_qc_panel_path', 'infection_intensity_mode', 'infection_intensity_qc', 'straightness_threshold', 'drop_straight_tracks', 'track_outlier_zscore', 'max_displacement',
'infection_pca_umap_search','infection_pca_umap_n_neighbors_grid','infection_pca_umap_min_dist_grid','infection_pca_pathogen_weight', 'infection_pca_log_intensity','infection_pca_tsne_search','infection_pca_tsne_perplexity_grid',
'infection_pca_tsne_learning_rate_grid', 'infection_pca_umap_n_neighbors','infection_pca_umap_min_dist','infection_pca_tsne_perplexity', 'infection_pca_min_silhouette','infection_pca_min_gt_separation','infection_pca_max_cells']
_organelle_all_settings = [
"organelle_morphology", "organelle_method", "organelle_diameter",
"organelle_mask_within_cells", "organelle_rolling_ball", "organelle_rolling_ball_radius", "organelle_clahe", "organelle_clahe_clip_limit",
"organelle_adaptive_block_size", "organelle_adaptive_offset",
"organelle_tophat_radius", "organelle_watershed_spots", "organelle_log_min_sigma", "organelle_log_max_sigma", "organelle_log_num_sigma", "organelle_log_threshold", "organelle_dog_sigma_low", "organelle_dog_sigma_high",
"organelle_ridge_filter", "organelle_ridge_sigmas", "organelle_skeletonize", "organelle_network_threshold", "organelle_hysteresis_low", "organelle_hysteresis_high",
"organelle_ring_sigma_inner", "organelle_ring_sigma_outer", "organelle_ring_min_prominence", "organelle_ring_fill_method",
"organelle_morph_radius", "organelle_fill_holes",
"organelle_model_name", "organelle_cellprob_threshold", "organelle_flow_threshold", "organelle_resample",
"organelle_unet_model_path", "organelle_unet_threshold",
"remove_background_organelle", "organelle_background", "organelle_signal_to_noise", "organelle_min_area", "organelle_max_area", "organelle_min_intensity", "organelle_max_intensity", "organelle_perimeter_fraction", "organelle_remove_border", "organelle_remove_border_objects",
"summarize_organelles_by",
]
_organelle_all_settings.insert(0, "organelle_type")
def _partition_organelle_settings(members):
"""Split the organelle settings into (basic, advanced), order preserved."""
from .organelle_types import is_basic
return ([k for k in members if is_basic(k)],
[k for k in members if not is_basic(k)])
organelle_basic_settings, organelle_advanced_settings = (
_partition_organelle_settings(_organelle_all_settings))
organelle_basic_settings.insert(0, NUMBER_OF_ORGANELLES)
categories = {
"Paths": ["src", "mask_src", "test_src", "test_mask_src", "save_path", "custom_model_path", "resume_checkpoint", "dataset", "model_path", "tar_path", "grna_csv", "row_csv", "column_csv", "metadata_files", "paired_data", "score_data", "count_data"],
"General": ["cell_mask_dim", "cytoplasm", "cell_chann_dim", "cell_channel", "nucleus_chann_dim", "nucleus_channel", "nucleus_mask_dim", "organelle_channel", "organelle_mask_dim", "organelle_chann_dim", "pathogen_mask_dim", "pathogen_chann_dim", "pathogen_channel", "segmentation_backend", "channels", "channel_dims", "normalize", "magnification", "metadata_type", "custom_regex", "experiment", "plot", "test_mode", "timelapse", "apply_model_to_dataset", "generate_training_dataset", "generate_full_dataset", "delete_intermediate", "uninfected", "object_filters", "real_object_classifier", "real_object_threshold"],
"Cellpose": ["channel_axis", "min_train_masks", "max_train_images",
"nimg_per_epoch", "nimg_test_per_epoch", "scale_range",
"save_every", "save_each", "base_model", "custom_model", "fill_in", "from_scratch", "n_epochs", "width_height", "target_size", "resample", "rescale", "CP_prob", "flow_threshold", "percentiles", "invert", "diameter", "grayscale", "Signal_to_noise", "resize", "target_height", "target_width", "plaque_model"],
"Cell": ["cell_model_name", "cell_diameter", "cell_background", "cell_signal_to_noise", "cell_cellprob_threshold", "cell_flow_threshold", "remove_background_cell", "adjust_cells", "cell_remove_border_objects", "cell_perimeter_fraction"],
"Nucleus": ["nucleus_model_name", "nucleus_diameter", "nucleus_background", "nucleus_signal_to_noise", "nucleus_cellprob_threshold", "nucleus_flow_threshold", "remove_background_nucleus", "nucleus_remove_border_objects", "nucleus_perimeter_fraction"],
"Pathogen": ["pathogen_model_name", "pathogen_diameter", "pathogen_background", "pathogen_signal_to_noise", "pathogen_cellprob_threshold", "pathogen_flow_threshold", "pathogen_model", "remove_background_pathogen", "pathogen_remove_border_objects", "pathogen_perimeter_fraction"],
"Organelle": organelle_basic_settings,
"Organelle advanced": organelle_advanced_settings,
"Host–Pathogen Analysis": ["hp_vacuole_table", "hp_vacuole_prefix",
"hp_reference_table", "hp_reference_prefix", "hp_marker_channels",
"hp_marker_thresholds", "hp_parasite_table", "hp_parasite_parent",
"hp_count_column"],
"Point Spread Function": ["psf_measurement_source", "psf_operation", "psf_source", "psf_objective",
"psf_path", "psf_image_sampling_um", "psf_kernel_sampling_um",
"psf_fwhm_um", "psf_iterations"],
"Spectral Unmixing α": ["unmix", "unmix_controls",
"unmix_background_percentile"],
"Self-Supervised Denoising α": ["n2v_denoise", "n2v_model",
"n2v_epochs"],
"Image Enhancement": ["enhance_background", "enhance_background_radius",
"enhance_background_scale",
"enhance_denoise", "enhance_denoise_strength",
"enhance_percentile_clip", "enhance_percentile_low",
"enhance_percentile_high",
"enhance_gamma", "enhance_log", "enhance_log_gain",
"enhance_sqrt",
"enhance_clahe", "enhance_clahe_tile", "enhance_clahe_clip",
"enhance_equalize",
"enhance_sharpen", "enhance_sharpen_radius",
"enhance_sharpen_amount"],
"Image Quality": ['image_qc_mode', 'image_qc_channels', 'image_qc_min_focus',
'image_qc_max_saturation', 'image_qc_saturation_level', 'image_qc_max_nonfinite',
'image_qc_classifier', 'image_qc_classifier_model',
'image_qc_classifier_labels', 'image_qc_classifier_threshold'],
"Cellpose 3": ["cellpose3_add_nucleus_channel", "cellpose3_size_model", "cellpose3_resample", "cellpose3_augment", "cellpose3_percentile_low", "cellpose3_percentile_high"],
"Segmentation QC": ["seg_qc", "seg_qc_min_objects", "seg_qc_count_ratio", "seg_qc_size_ratio", "seg_qc_border_fraction", "seg_qc_outlier_mad", "seg_qc_outlier_fraction", "seg_qc_foreground_fraction", "seg_qc_split_ratio", "seg_qc_min_diameter", "seg_qc_tiny_fraction", "seg_qc_max_object_fraction", "seg_qc_plate_fail_fraction"],
"Segmentation Robustness α": ["robustness_report", "robustness_fields",
"robustness_crop", "robustness_diameter_factors",
"robustness_flow_thresholds",
"robustness_cellprob_thresholds",
"robustness_enhancement", "robustness_tolerance"],
"Timelapse": timelapse_settings,
"Measurements": ["save_measurements", "calculate_correlation", "spatial_measurements", "spatial_neighbor_radius", "bystander_measurements", "bystander_reach_in_diameters", "homogeneity", "homogeneity_distances", "radial_dist", "distance_gaussian_sigma", "tables", "parasite_table", "compartment", "channel_of_interest", "measurement", "filter_by", "exclude", "cell_min_size", "cytoplasm_min_size", "nucleus_min_size", "pathogen_min_size", "cell_max_size", "nucleus_max_size", "pathogen_max_size", "object_distances", "object_distance_maxima", "object_distance_intensity", "merge_edge_pathogen_cells", "cell_size_range", "cell_intensity_range", "nucleus_size_range", "nucleus_intensity_range", "pathogen_size_range", "pathogen_intensity_range", "cells_per_well", "target_intensity_min", "nuclei_limit", "pathogen_limit", "remove_highly_correlated", "remove_highly_correlated_features", "remove_low_variance_features"],
"Illumination Correction": ["illumination_correction", "illumination_model", "illumination_estimator", "illumination_degree", "illumination_dark", "illumination_per_plate", "illumination_max_fields", "illumination_qc", "illumination_on_missing", "illumination_vendor_profile", "illumination_vendor_channel_map"],
"Object Crops": ["save_png", "crop_mode", "png_size", "png_channel_mapping", "png_dims", "dialate_pngs", "dialate_png_ratios", "use_bounding_box", "normalize_by", "save_arrays"],
"Plate Layout & Controls": ["plaque_mode", "figure_detector", "figure_imgsz", "figure_confidence", "figure_read_text", "confirm_annotations", "text_reach_above", "text_reach_left", "text_reach_below", "text_use_above", "text_use_left", "text_use_below", "text_panel_reach", "text_min_confidence", "text_ignore", "text_order", "text_separator", "text_reread", "text_reread_scale", "well_detection", "well_confidence", "well_pad", "plate_format", "well_diameter_mm", "plaque_pixels_per_um", "plaque_formation_hours", "plaque_estimate_growth", "plaque_growth_reference_um", "plaque_growth_reference_hours", "plateID", "plate", "cell_types", "cell_plate_metadata", "cells", "cell_loc", "pathogen_types", "pathogen_plate_metadata", "pathogens", "pathogen_loc", "treatments", "treatment_plate_metadata", "treatment_loc", "location_column", "group_column", "level", "change_plate", "positive_control_id", "negative_control_id", "exclude_grnas", "positive_control_wells", "negative_control_wells", "mixed_control_wells", "nontargeting_control_grnas", "pos", "neg", "mix", "exclude_conditions", "exclude_rows", "filter_column", "filter_value", "target", "batch_correction", "batch_column", "batch_control_column", "batch_control_values", "batch_covariate_column", "batch_combat_mean_only", "batch_min_samples", "batch_missing_control"],
"Training Classes": ["dataset_mode", "classes", "class_folder_names", "class_metadata", "metadata_item_1_name", "metadata_item_1_value", "metadata_item_2_name", "metadata_item_2_value", "annotation_column", "annotated_classes", "write_random_annotation_column"],
"Computer Vision Data Source": ["image_source", "image_size", "size", "train_channels", "stream_method", "object_array", "channel_arrays", "bounding_box", "crop_shape", "sample", "test_split", "val_split", "balance_to_smallest", "augment",
"crop_source", "file_metadata", "file_type", "coordinate_columns"],
"Computer Vision Model": ["model_type", "model_name", "init_weights", ],
"Computer Vision Training": ["train", "test", "epochs", "learning_rate", "optimizer_type", "schedule", "loss_type", "label_smoothing", "focal_gamma", "focal_alpha", "logit_adjust_tau", "class_balance", "amsgrad", "mixed_precision", "gradient_accumulation_steps", "early_stopping_patience", "pin_memory", "intermedeate_save", "tensorboard", "random_seed",
"n_top_examples"],
"Computer Vision Optimization and Regularization": ["use_checkpoint", "dropout_rate", "weight_decay"],
"Test-time augmentation": ['tta_enabled', 'tta_rotations', 'tta_horizontal_flip',
'tta_vertical_flip', 'tta_aggregation', 'tta_min_agreement', 'tta_max_std'],
"Model Evaluation": ["cross_validation_enabled", "cross_validation_folds",
"cv_group_by", "holdout_plate", "nested_cv_inner_folds"],
"Evaluation Reports": ["classifier_evaluation", "evaluation_calibration",
"evaluation_bins", "score_threshold",
"score_column"],
"Leakage Audit": ["evaluation_fail_on_leakage", "leakage_audit_train_test",
"leakage_hash_content", "leakage_require_identity"],
"Machine Learning Model and Features": ["model_type_ml", "n_estimators", "test_size", "cross_validation", "reg_lambda", "reg_alpha", "prune_features", "top_features", "n_repeats", ],
"Embedding & Clustering": ["reduction_method", "n_neighbors", "min_dist", "metric", "tsne_perplexity", "tsne_learning_rate", "tsne_early_exaggeration", "tsne_max_iter", "pca_whiten", "pca_svd_solver", "isomap_n_neighbors", "isomap_path_method", "spectral_affinity", "spectral_n_neighbors", "log_data", "embedding_by_controls", "col_to_compare", "resnet_features", "clustering", "eps", "min_samples", "remove_cluster_noise", "analyze_clusters"],
"Regression: Response": [
"dependent_variable", "invert_dependent_variable",
"count_grna_column", "count_value_column",
"independent_variable_layout", "wide_predictor_columns",
"model_data_layout",
"analysis_unit", "agg_type", "transform",
],
"Regression: Model": [
"inference", "analysis_mode", "regression_type", "regression_backend",
"intercept", "intercept_value",
"model_plate_position", "random_row_column_effects", "cov_type",
],
"Regression: Model Tuning": [
"alpha", "l1_ratio", "quantile", "huber_t", "hinge_threshold",
"spline_knots", "spline_degree",
"hinge_n_boot", "lasso_n_boot", "lasso_selection_threshold",
"group_lasso_lambda",
],
"Regression: Permutation Test": [
"grna_statistic",
"guide_min_wells", "guide_primary_min_wells", "guide_permutations",
"guide_permutation_seed", "guide_permutation_block",
"guide_nuisance_columns", "guide_presence_threshold",
"guide_permutation_batch_size",
],
"Regression: Significance": [
"multiple_testing_method", "fdr_alpha", "threshold_method",
"threshold_multiplier", "annotation_source",
"p_threshold_alpha", "p_threshold_kind",
"rra_alpha", "rra_permutations",
],
"Regression: Quality Filters": [
"min_cells_per_well", "min_observations_per_hit", "fraction_threshold",
"calibrate_fraction_threshold",
"normalise_fraction",
"target_unique_count", "tolerance", "outlier_detection", "other",
],
"Regression: Diagnostics": ["regression_qc"],
"Activation Maps": ["smoothgrad_samples", "smoothgrad_sigma", "occlusion_window", "occlusion_stride", "ig_steps", "ig_baseline", "attribution_steps", "attribution_baseline", "sanity_check", "object_type", "cam_type", "target_layer", "overlay", "correlation", "manders_thresholds", "normalize_input", "counterfactuals", "counterfactual_crops", "counterfactual_epochs", "counterfactual_condition", "counterfactual_target", "counterfactual_generator"],
"Sequencing": ["mode", "single_direction", "target_sequence", "regex", "offset_start", "window_length", "barcode_mismatches", "chunk_size", "fill_na", "save_h5", "comp_type", "comp_level"],
"Plot": ["cmap", "figuresize", "black_background", "save_figure", "log_x", "log_y", "x_lim", "y_lims", "examples_to_plot", "plot_control", "plot_nr", "nr_imgs", "um_per_pixel", "image_nr", "dot_size", "point_color", "point_alpha", "outline_width", "umap_canvas_width", "umap_sidebar_width", "img_zoom", "row_limit", "color_by", "plot_images", "remove_image_canvas", "plot_points", "plot_outlines", "smooth_lines", "plot_by_cluster", "plot_cluster_grids", "heatmap_feature", "grouping", "min_max"],
"Replication Assay": [
'replication_method',
"vacuole_key", "vacuole_link_distance", "vacuole_link_factor",
"parasite_count_column", "max_parasites_per_vacuole",
"require_host_cell", "non_power_of_two_warn",
],
"Endodyogeny Size Proxy (Legacy)": [
"class_column", "group_by_class", "um_per_px",
"min_area_bin", "max_area", "max_bins",
],
"Invasion Assay": [
"outside_channel", "total_channel",
"intensity_statistic", "background_correction",
"outside_threshold_method", "outside_threshold", "stain_baseline_wells",
"control_quantile", "min_control_objects", "min_objects_for_threshold",
"min_objects_for_bimodality", "bimodality_cutoff",
"threshold_agreement_tolerance", "threshold_sensitivity",
"inflation_warn", "min_parasites_per_well",
"min_parasite_area", "max_parasite_area", "min_total_intensity",
"extracellular_class",
"seed_wells_from_cells",
"qc_plot_max_panels",
],
"Advanced": ["resume", "strict_errors", "max_failure_rate", "queue_by_uncertainty", "queue_measure", "queue_diversity", "queue_limit", "dry_run", "watch_folder", "watch_pipeline", "watch_normalization_pool", "watch_measure_settings", "watch_classify_settings", "watch_settle_seconds", "watch_poll_seconds", "watch_idle_minutes", "microscope_feedback", "microscope_driver", "microscope_simulated_folder", "microscope_positions", "microscope_stage_transform", "microscope_event_table", "microscope_event_query", "microscope_max_events", "microscope_timepoints", "microscope_interval_seconds", "cloud_anonymous", "cloud_profile", "cloud_endpoint", "cloud_cache", "cloud_wells", "cloud_fields", "cloud_level", "cloud_results", "verbose", "n_jobs", "ram_guard", "gpu", "mask_parallel", "mask_gpu_indices", "batch_size", "test_images", "random_test", "test_nr", "preprocess", "masks", "remove_background", "background", "backgrounds", "lower_percentile", "randomize", "batch_fields", "pipeline_style", "keep_intermediate", "keep_original_images", "save_original_images", "keep_npz", "diameter_estimate_n_fields", "shuffle", "save", "filter", "merge_pathogens", "consolidate", ],
"3D Settings (Beta)": [
"z_stack", "z_segmentation_mode", "z_axis", "z_projection",
"anisotropy", "voxel_size_z_um", "voxel_size_xy_um",
"stitch_threshold",
],
"4D Settings (Beta)": [
"t_stack", "t_axis_order", "t_axis", "frame_interval_s",
"t_track_backend", "t_link_threshold", "t_max_displacement_px",
"t_max_displacement_um", "t_project_for_tracking",
],
"Confluency α": [
"confluency", "confluency_source", "confluency_channel",
"confluency_window", "confluency_qc_threshold",
],
"Colony Counting α": [
"colony_counting", "colony_dilution", "colony_plated_volume_ul",
"colony_too_many", "colony_too_few", "colony_polarity",
"colony_threshold", "colony_min_area_px", "colony_detector",
],
"Bleach Correction α": [
"bleach_correction",
],
"GPU Measurement α": [
"measure_gpu",
],
"Measurement Backend α": [
"measurement_backend", "measurement_backend_target",
],
"Profiling α": [
"profiling", "profiling_metadata", "profiling_treatment_column",
"profiling_negative_control", "profiling_normalization",
"profiling_feature_selection", "profiling_correlation_threshold",
"profiling_phenotype_column", "profiling_databases",
],
"Cell Cycle α": [
"cell_cycle", "cell_cycle_method", "cell_cycle_channel",
"cell_cycle_gates", "cell_cycle_mitotic_ratio",
"cell_cycle_fucci_channels", "cell_cycle_labels", "cell_cycle_model",
"cell_cycle_epochs",
],
"Wound Closure α": [
"wound_closure", "wound_source", "wound_channel", "wound_window",
"wound_threshold", "wound_hours_per_frame", "wound_conditions",
],
"Intensity Calibration α": [
"intensity_calibration", "intensity_calibration_wells",
"intensity_calibration_statistic", "intensity_calibration_offset",
],
"Plate Barcode Linkage α": [
"plate_barcode_source", "plate_barcodes", "plate_barcode_column",
"plate_barcode_token_env",
],
"Time To Event α": [
"time_to_event", "time_to_event_object", "time_to_event_mode",
"time_to_event_column", "time_to_event_threshold",
"time_to_event_persist", "time_to_event_origin",
"time_to_event_min_frames", "time_to_event_hours_per_frame",
"time_to_event_group", "time_to_event_conditions",
"time_to_event_reference", "time_to_event_covariates",
],
"Viability α": [
"viability", "viability_dead_channel", "viability_live_channel",
"viability_thresholds", "viability_negative_wells",
"viability_positive_wells", "viability_plate_map",
],
"CellProfiler α": [
"cellprofiler_pipeline",
],
"Motility (beta)": motility_settings,
"Motility Advanced (beta)": motility_advanced_settings,
}
#: The umbrella every advanced family hangs under.
#:
#: NOT "Advanced". That heading already exists and holds the RUN's advanced
#: knobs -- resume, n_jobs, batch_size, keep_intermediate -- which are about
#: how the job executes, not about what is done to an object. Two headings
#: three letters apart would be two places to look for the same thing.
ADVANCED_UMBRELLA = 'Advanced settings'
#: The per-object image preprocessing heading.
#:
#: NOT "Image preprocessing", which mask and timelapse already draw for the
#: WHOLE-IMAGE steps -- normalize, upscale, denoise, lower_percentile. The
#: category-help table is keyed on the heading's exact text, so a second
#: heading spelled the same way would silently serve the wrong blurb; that
#: is the half of the "Computer Vision --" prefix precedent that broke every
#: tooltip lookup for its groups. The parenthetical is what keeps the two
#: keys apart, and it says the true difference: this one is applied per
#: object channel.
PER_OBJECT_PREPROCESSING = 'Image preprocessing (per object)'
#: ``(heading, key suffixes, key prefixes)`` per advanced family.
#:
#: A family is matched on the SUFFIX after the object name (`cell_min_size`)
#: and on the PREFIX before it (`remove_background_cell`), because spaCR
#: spells the same relationship both ways round and a family that could only
#: see one of them would split a decision in half.
_ADVANCED_FAMILIES = (
(PER_OBJECT_PREPROCESSING, (
"background", "signal_to_noise",
"rolling_ball", "rolling_ball_radius", "clahe", "clahe_clip_limit",
), (
"remove_background",
)),
("Object filtration", (
"min_size", "max_size", "min_area", "max_area",
"min_intensity", "max_intensity",
"perimeter_fraction", "remove_border", "remove_border_objects",
), ()),
)
_FAMILY_SHARED_KEYS = {"Object filtration": ("object_filters",
"real_object_classifier",
"real_object_threshold")}
"""Keys a family heading takes whole rather than one per object.
``object_filters`` (item 511) holds every object type's filter list in one
mapping, so it has no object prefix for the suffix match to find, and it
belongs beside the per-object area and intensity bounds it extends. The
real / not-real classifier and its threshold remove objects after
detection too, for every object type at once.
"""
#: Which heading each category nests under when the panel can draw a tree.
#:
#: DECLARED BESIDE THE CATEGORIES, NOT BY RENAMING THEM. Encoding the parent
#: in the heading text -- "Advanced settings / Object filtration" -- is the
#: prefix trick, and the prefix trick is what broke every tooltip lookup for
#: the Computer Vision groups, because the help table is keyed on the exact
#: heading text. A separate table leaves every existing name, blurb and
#: dependency untouched.
CATEGORY_PARENTS = {
heading: ADVANCED_UMBRELLA for heading, _suffixes, _prefixes
in _ADVANCED_FAMILIES
}
#: Categories the regroup leaves alone.
#:
#: "Measurements" holds `cell_min_size` and its siblings by an EARLIER
#: DELIBERATE DECISION: they are measurement filters that only measure_crop
#: sets, and filing them under Cell / Nucleus / Pathogen gave the Measure
#: module three headings holding one size field each and no segmentation.
#: Pulling them into a shared filtration heading would undo that and put the
#: same three near-empty headings back by another route.
_ADVANCED_REGROUP_EXEMPT = ("Measurements",)
#: Object order within each family heading, so the same decision for four
#: objects reads as one block rather than four scattered rows. It is also the
#: order the per-object sub-headings are drawn in.
ADVANCED_OBJECT_ORDER = (
"cell", "nucleus", "pathogen", "cytoplasm", *ORGANELLE_SLOT_ROLES)
_ADVANCED_OBJECT_ORDER = ADVANCED_OBJECT_ORDER
[docs]
def advanced_object_of(key):
"""Which object a family member belongs to, or None.
Both spellings are understood -- `cell_min_size` and
`remove_background_cell` -- because the sub-heading a key is drawn under
has to be the object it acts on whichever way round spaCR happens to
name it.
:param key: a settings key.
:returns: the object name from :data:`ADVANCED_OBJECT_ORDER`, or None
when the key names no object.
"""
text = str(key)
for obj in ADVANCED_OBJECT_ORDER:
if text.startswith(f"{obj}_") or text.endswith(f"_{obj}"):
return obj
return None
def _advanced_lookup_sets(table):
"""The two membership sets every family in ``table`` is matched against.
:param table: the category table being regrouped.
:returns: ``(spoken_for, filed)`` -- the keys an exempt category has
already claimed, and every key the table files anywhere.
SPLIT OUT BECAUSE `filed` IS THE WHOLE TABLE. spaCR files 37,165 keys
across 54 categories, and building that set is most of the cost of
matching one family. It was rebuilt for every family, so the regroup paid
for it six times over to get six identical answers.
"""
spoken_for = {k for c in _ADVANCED_REGROUP_EXEMPT
for k in table.get(c, ())}
filed = {key for members in table.values() for key in members}
return spoken_for, filed
def _prefixed_family_key(prefix, obj):
"""The key a prefix-form family names for one object.
``remove_background_<object>`` for every object but an organelle slot
after the first, whose background switch is numbered as the user counts
slots (``remove_background_organelle_2``, see
:func:`spacr.organelle_types._background_switch_key`).
"""
if f"{prefix}_" == "remove_background_" and obj.startswith("organelle"):
try:
return _background_switch_key(obj)
except ValueError:
pass
return f"{prefix}_{obj}"
def _advanced_family_members(table, family_suffixes, family_prefixes=()):
"""Keys belonging to one family, ordered by object then by suffix.
Matched on the SUFFIX after the object prefix, not by substring: `area`
would otherwise pull in `area_multiplier` twice and miss nothing useful.
A PREFIX form is matched too -- `remove_background_cell` puts the object
last -- so a family is not split in half by spaCR's own inconsistency
about which end the object name goes on.
"""
spoken_for, filed = _advanced_lookup_sets(table)
found = []
seen = set()
for obj in ADVANCED_OBJECT_ORDER:
candidates = [f"{obj}_{suffix}" for suffix in family_suffixes]
candidates += [_prefixed_family_key(prefix, obj)
for prefix in family_prefixes]
for key in candidates:
if key in spoken_for or key in seen:
continue
if key in filed:
found.append(key)
seen.add(key)
return found
def _regroup_advanced(table):
"""Move the shared families out of the per-object categories.
MOVED, NOT HIDDEN, and never renamed: a settings CSV names keys, not
headings, so a file written before this loads and means exactly what it
meant. `test_the_regroup_does_not_change_which_keys_a_module_offers`
is the guard.
"""
out = dict(table)
by_family = [
(heading, _advanced_family_members(out, suffixes, prefixes))
for heading, suffixes, prefixes in _ADVANCED_FAMILIES
]
filed_anywhere = {key for keys in out.values() for key in keys}
by_family = [
(heading, members + [key for key in _FAMILY_SHARED_KEYS.get(heading, ())
if key in filed_anywhere and key not in members])
for heading, members in by_family
]
moved = set()
for _heading, members in by_family:
moved.update(members)
for category, keys in list(out.items()):
if category in _ADVANCED_REGROUP_EXEMPT:
continue
if not any(k in moved for k in keys):
continue
out[category] = [k for k in keys if k not in moved]
family_headings = {heading for heading, _s, _p in _ADVANCED_FAMILIES}
for heading, members in by_family:
if members:
out[heading] = members
return {k: v for k, v in out.items() if v or k in family_headings}
_organelle_basic_slots = [key for key in organelle_basic_settings
if key.startswith('organelle_')]
_organelle_advanced_slots = [key for key in organelle_advanced_settings
if key.startswith('organelle_')]
for _role in ORGANELLE_SLOT_ROLES[1:]:
categories['Organelle'].extend(
_organelle_slot_key(key, _role) for key in _organelle_basic_slots)
categories['Organelle advanced'].extend(
_organelle_slot_key(key, _role) for key in _organelle_advanced_slots)
categories['Organelle advanced'].append(_background_switch_key(_role))
for _suffix in ('channel', 'mask_dim', 'chann_dim'):
_key = f'{_role}_{_suffix}'
categories['General'].append(_key)
del _organelle_basic_slots, _organelle_advanced_slots
_regrouped_categories = _regroup_advanced(categories)
categories.clear()
categories.update(_regrouped_categories)
del _regrouped_categories
category_dependencies = {
'timelapse': ['Timelapse'],
'motility_analysis': ['Motility (beta)', 'Motility Advanced (beta)'],
}
category_group_dependencies = {}
category_integer_dependencies = {
('cell_channel', 'cell_mask_dim'): ['Cell'],
('nucleus_channel', 'nucleus_mask_dim'): ['Nucleus'],
('pathogen_channel', 'pathogen_mask_dim'): ['Pathogen'],
tuple(key for role in ORGANELLE_SLOT_ROLES
for key in (f'{role}_channel', f'{role}_mask_dim')): [
'Organelle', 'Organelle advanced'],
}
setting_dependencies = {}
[docs]
def get_setting_dependencies():
"""Return reviewed rules for settings that currently have no effect.
Estimator rules are generated from ``ml.REGRESSION_SETTINGS_USED`` -- the
same inventory that rejects unused knobs at run time -- so the GUI cannot
drift into enabling a setting the selected backend refuses.
"""
if setting_dependencies:
return setting_dependencies
from .regression_spec import REGRESSION_SETTINGS_USED
def rule(sources, predicate, reason):
"""One dependency rule: its sources, its predicate and its reason."""
return {
'sources': tuple(sources),
'predicate': predicate,
'reason': reason,
}
def _combined(existing, sources, predicate, reason):
"""A second reason a setting can be inapplicable, ANDed with the first.
One setting can be dead for more than one reason at once -- a
permutation control is dead under parametric inference AND dead on a
single plate -- and this dict holds one entry per setting. Replacing
the entry drops the earlier reason silently, which is how a field
ends up enabled among its greyed siblings.
Applicable only when BOTH agree. The reason shown is the one that
actually fired, so the user is told why THIS control is off rather
than being given the other rule's explanation.
"""
if not existing:
return rule(sources, predicate, reason)
def _predicate(settings, context):
"""True only when BOTH the existing rule and the new one hold."""
return bool(existing['predicate'](settings, context)) and \
bool(predicate(settings, context))
def _reason(settings, context):
"""The reason from whichever rule is the one failing.
The EXISTING rule is asked first, so a setting gated by two conditions
explains the one that has been true for longer rather than the one added
most recently.
"""
if not existing['predicate'](settings, context):
return existing['reason'](settings, context)
return reason(settings, context)
return rule(tuple(existing['sources']) + tuple(sources),
_predicate, _reason)
owned = sorted({key for keys in REGRESSION_SETTINGS_USED.values()
for key in keys})
for key in owned:
setting_dependencies[key] = rule(
('regression_type',),
lambda settings, context, setting=key: setting in set(
REGRESSION_SETTINGS_USED.get(
str(settings.get('regression_type') or '').lower(), ())),
lambda settings, context, setting=key: (
f"{setting} is not read when regression_type is "
f"{settings.get('regression_type')!r}. Choose a family whose "
"documented settings include it. The value is kept and saved."),
)
guide_keys = (
'grna_statistic',
'guide_min_wells', 'guide_primary_min_wells', 'guide_permutations',
'guide_permutation_seed', 'guide_permutation_block',
'guide_nuisance_columns', 'guide_presence_threshold',
'guide_permutation_batch_size',
)
def permutation_active(settings, _context):
"""Whether the RUN would permute, decided the way the run decides.
THE SAME TRANSLATION `set_default_analysis_settings` APPLIES, and it
has to be: `inference` is the readable front end and it OVERWRITES
`analysis_mode` at run time, so a panel that read the stale
`analysis_mode` answered a question the fit does not ask. Measured on
the regression panel: choosing inference='parametric' left all eight
guide-permutation controls enabled, because `analysis_mode` still held
the value the previous inference had selected -- a control offered for
a step that is not going to run.
`auto` is the one case where `analysis_mode` still decides here.
Its real resolution counts guides and wells, which this cannot see
(`spacr.ml.resolve_auto_inference` does it once the CSVs are read), so
the permutation controls stay ENABLED under 'auto' -- greying a
control the run may well use is the worse error of the two.
"""
inference = str(settings.get('inference') or 'auto').strip().lower()
selected = INFERENCE_MODES.get(inference, None)
if selected is not None:
return selected == 'guide_permutation'
mode = str(settings.get('analysis_mode') or '').lower()
return inference == 'auto' or mode == 'guide_permutation'
def permutation_is_certain(settings) -> bool:
"""Whether the run WILL permute -- not merely whether it might.
NOT `not permutation_active(...)`, which is the mistake this replaces
and which my own comment below warned about before I made it.
`permutation_active` answers True under 'auto' ON PURPOSE, so the
permutation controls stay live while the resolution is unknown.
Negating it therefore greys the MODEL controls under 'auto' too --
and 'auto' may well resolve to regression, in which case those are
exactly the settings the run reads.
Certain means `inference` names the permutation path outright.
"""
inference = str(settings.get('inference') or 'auto').strip().lower()
selected = INFERENCE_MODES.get(inference, None)
if selected is not None:
return selected == 'guide_permutation'
return (inference != 'auto'
and str(settings.get('analysis_mode') or '').lower()
== 'guide_permutation')
for key in ('regression_type', 'regression_backend', 'cov_type',
'intercept', 'intercept_value',
'model_plate_position', 'random_row_column_effects'):
setting_dependencies[key] = _combined(
setting_dependencies.get(key),
('inference', 'analysis_mode'),
lambda settings, context: not permutation_is_certain(settings),
lambda settings, context, setting=key: (
f"{setting} is not read under nonparametric inference: the "
f"guide-permutation path tests each guide on its own with "
f"permutations and fits no model, so no "
f"regression family is chosen. The value is kept and saved."),
)
for key in guide_keys:
setting_dependencies[key] = rule(
('inference', 'analysis_mode'), permutation_active,
lambda settings, context, setting=key: (
f"{setting} is read only by nonparametric guide permutation "
f"inference (currently {settings.get('inference')!r}). The "
"value is kept and saved."),
)
setting_dependencies['transform'] = _combined(
setting_dependencies.get('transform'),
('regression_type',),
lambda settings, _context: not (
str(settings.get('regression_type') or '').lower() == 'glm'
and str(settings.get('transform') or '').strip().lower()
in ('log', 'logit')),
lambda settings, _context: (
f"transform={settings.get('transform')!r} is a link, and a glm "
"fits the response as measured so the family's own link does "
"that job -- applying both would fit logit(log(y)). To fit the "
"transformed response instead, use regression_type='ols'. The "
"value is kept and saved."),
)
setting_dependencies['intercept_value'] = _combined(
setting_dependencies.get('intercept_value'),
('intercept',),
lambda settings, _context: str(
settings.get('intercept') or 'fitted').strip().lower() == 'value',
lambda settings, _context: (
f"intercept={str(settings.get('intercept') or 'fitted')!r} does "
"not read a pinned number: 'fitted' estimates the intercept, "
"'zero' fits through the origin and 'control' pins it at the "
"negative controls. Choose intercept='value' to use this field. "
"The value is kept and saved."),
)
setting_dependencies.setdefault('group_lasso_lambda', rule(
('regression_type',),
lambda settings, context: str(
settings.get('regression_type') or '').lower() == 'group_lasso',
lambda settings, context: (
f"group_lasso_lambda is not read when regression_type is "
f"{settings.get('regression_type')!r}; it is the group lasso's "
f"penalty weight. The value is kept and saved."),
))
setting_dependencies['analysis_mode'] = rule(
('inference',),
lambda settings, context: str(
settings.get('inference') or 'auto').lower() == 'auto',
lambda settings, context: (
f"analysis_mode is set for you by inference="
f"{settings.get('inference')!r}, which selects "
f"{INFERENCE_MODES.get(str(settings.get('inference') or '').lower()) or 'regression'!r}. "
f"Choose inference='auto' to pick the mode by hand. The value is "
f"kept and saved."),
)
def _is_nonparametric(settings):
"""Whether inference is the nonparametric permutation test."""
return str(settings.get('inference') or '').lower() == 'nonparametric'
def _level_is_read(settings, _context):
"""Whether the level setting is read at all under these settings.
THE PERMUTATION TEST READS IT. It fits no model, so `regression_type`
says nothing about it -- greying the control on the parametric answer
left the nonparametric side with no way to ask for genes at all.
"""
if _is_nonparametric(settings):
return True
if settings.get('random_row_column_effects', False):
return False
return str(settings.get('regression_type') or '').lower() != 'mixed'
def _level_reason(settings, _context):
"""Why the level control is greyed, naming the setting responsible."""
if settings.get('random_row_column_effects', False) and \
str(settings.get('regression_type') or '').lower() != 'mixed':
return (
"level is not read because random_row_column_effects=True "
"fits a mixed model whatever regression_type says, and a "
"mixed model already covers both levels at once. Untick it, "
"or choose the level on a fixed-effects family. The value is "
"kept and saved.")
return (
"level is not read when regression_type is 'mixed': the mixed "
"model fits the gene as a fixed effect and each guide as a random "
"effect NESTED inside its gene, so it answers both levels at once "
"and there is no single one to pick. Choose any other family to "
"fit one level at a time. The value is kept and saved.")
setting_dependencies['level'] = rule(
('regression_type', 'random_row_column_effects', 'inference'),
_level_is_read, _level_reason)
_cell_unit = lambda settings, context: str(
settings.get('analysis_unit') or 'well').lower() == 'cell'
setting_dependencies['inference'] = rule(
('analysis_unit',),
lambda settings, context: not _cell_unit(settings, context),
lambda settings, context: (
"analysis_unit='cell' keeps one row per object, and the "
"permutation test needs one row per well -- so the inference is "
"'parametric' here. Set analysis_unit='well' with an agg_type "
"to choose it. The value is kept and saved."),
)
setting_dependencies['analysis_mode'] = _combined(
setting_dependencies.get('analysis_mode'),
('analysis_unit',),
lambda settings, context: not _cell_unit(settings, context),
lambda settings, context: (
"analysis_unit='cell' gives one row per object, which only the "
"model can read -- so analysis_mode is 'regression' here."),
)
setting_dependencies['agg_type'] = rule(
('analysis_unit',),
lambda settings, context: settings.get('analysis_unit') == 'well',
lambda settings, context: (
"agg_type is read only when analysis_unit is 'well' "
f"(currently {settings.get('analysis_unit')!r}). The value is "
"kept and saved."),
)
batch_active = lambda settings, context: str(
settings.get('batch_correction') or 'none').lower() != 'none'
for key in ('batch_column', 'batch_min_samples', 'batch_missing_control'):
setting_dependencies[key] = rule(
('batch_correction',), batch_active,
lambda settings, context, setting=key: (
f"{setting} is read only when batch_correction is enabled "
f"(currently {settings.get('batch_correction')!r}). The value "
"is kept and saved."),
)
for key in ('batch_control_column', 'batch_control_values'):
setting_dependencies[key] = rule(
('batch_correction',),
lambda settings, context: str(
settings.get('batch_correction') or '').lower()
== 'control_center',
lambda settings, context, setting=key: (
f"{setting} is read only when batch_correction is "
f"'control_center' (currently "
f"{settings.get('batch_correction')!r}). The value is kept."),
)
for key in ('batch_covariate_column', 'batch_combat_mean_only'):
setting_dependencies[key] = rule(
('batch_correction',),
lambda settings, context: str(
settings.get('batch_correction') or '').lower() == 'combat',
lambda settings, context, setting=key: (
f"{setting} is read only when batch_correction is 'combat' "
f"(currently {settings.get('batch_correction')!r}). The value "
"is kept."),
)
_existing = setting_dependencies.get('guide_permutation_block')
setting_dependencies['guide_permutation_block'] = _combined(
_existing,
('paired_data', 'score_data', 'count_data'),
lambda settings, context: (
(context or {}).get('plate_count') is None
or (context or {}).get('plate_count') != 1),
lambda settings, context: (
"guide_permutation_block names the column permutations are "
"blocked within, and residuals are never shuffled between its "
"levels. The loaded inputs hold one plate, so blocking on the "
"plate is the whole dataset and constrains nothing. The value is "
"kept and saved."),
)
for _key in ('batch_correction', 'batch_column', 'batch_control_column',
'batch_control_values', 'batch_covariate_column',
'batch_combat_mean_only', 'batch_min_samples',
'batch_missing_control'):
setting_dependencies[_key] = _combined(
setting_dependencies.get(_key),
('paired_data', 'score_data', 'count_data'),
lambda settings, context: (
(context or {}).get('plate_count') is None
or (context or {}).get('plate_count') != 1),
lambda settings, context, key=_key: (
f"{key} configures correction BETWEEN batches, and the loaded "
f"inputs hold one plate. There is nothing between batches to "
f"remove: batch_correction='none' gives an identical result. "
f"The value is kept and saved."),
)
from ._stream_selection import METHOD_SETTINGS as _METHOD_SETTINGS
def _streaming(settings) -> bool:
"""Whether the run reads images from a stream rather than from disk."""
return _canonical_image_source(
settings.get('image_source', settings.get('crop_source'))
) == 'stream_images'
_stream_only = tuple(dict.fromkeys(
['stream_method',
*(key for keys in _METHOD_SETTINGS.values() for key in keys)]))
for _key in _stream_only:
setting_dependencies[_key] = _combined(
setting_dependencies.get(_key),
('image_source',),
lambda settings, context: _streaming(settings),
lambda settings, context, key=_key: (
f"{key} is read only while streaming, and image_source is "
f"'load_images', which reads crops that already exist. The "
f"value is kept and saved."),
)
for _key in _stream_only[1:]:
setting_dependencies[_key] = _combined(
setting_dependencies.get(_key),
('stream_method',),
lambda settings, context, key=_key: key in _METHOD_SETTINGS.get(
str(settings.get('stream_method') or '').strip().lower(),
_stream_only[1:]),
lambda settings, context, key=_key: (
f"{key} is not read when stream_method is "
f"{settings.get('stream_method')!r}. The value is kept and "
f"saved."),
)
setting_dependencies['custom_regex'] = _combined(
setting_dependencies.get('custom_regex'),
('metadata_type',),
lambda settings, context: str(
settings.get('metadata_type') or '').strip().lower()
in ('custom', 'auto'),
lambda settings, context: (
f"custom_regex is only read when metadata_type is 'custom', or "
f"'auto' with a regex supplied. It is "
f"{settings.get('metadata_type')!r}. The value is kept and saved."),
)
setting_dependencies['organelle_model_name'] = _combined(
setting_dependencies.get('organelle_model_name'),
('organelle_method',),
lambda settings, context: str(
settings.get('organelle_method') or '').strip().lower()
== 'cellpose',
lambda settings, context: (
f"organelle_model_name is only read when organelle_method is "
f"'cellpose'. It is {settings.get('organelle_method')!r}, which "
f"segments without a checkpoint. The value is kept and saved."),
)
setting_dependencies['illumination_vendor_channel_map'] = rule(
('illumination_correction', 'illumination_vendor_profile', 'illumination_model'),
lambda settings, context: (
bool(settings.get('illumination_correction', False))
and bool(str(settings.get('illumination_vendor_profile') or '').strip())
and not str(settings.get('illumination_model') or '').strip()),
lambda settings, context: (
f"illumination_vendor_channel_map is only read when "
f"illumination_correction is on, illumination_vendor_profile names a "
f"profile and illumination_model is empty. They are "
f"{settings.get('illumination_correction', False)!r}, "
f"{settings.get('illumination_vendor_profile')!r} and "
f"{settings.get('illumination_model')!r}. The value is kept and saved."),
)
setting_dependencies['bleach_correction'] = rule(
('timelapse',),
lambda settings, context: bool(settings.get('timelapse', False)),
lambda settings, context: (
"Bleach correction is only used for timelapse runs. The value is kept and saved."),
)
for _key in categories.get("Time To Event α", ()):
setting_dependencies[_key] = _combined(
setting_dependencies.get(_key),
('timelapse',),
lambda settings, context: bool(settings.get('timelapse', False)),
lambda settings, context, key=_key: (
f"{key} follows tracked objects, so it is only read when "
f"timelapse is on. The value is kept and saved."),
)
return setting_dependencies
category_value_dependencies = {
'organelle_method': {},
}
category_keys = list(categories.keys())
[docs]
def parse_list(value):
"""Parse a string literal into a homogeneous list of scalars.
Accepts Python-list or tuple literals and rejects mixed-type contents.
Single-element tuples are returned as one-element lists.
Lives here, beside its only caller :func:`check_settings`, rather than in
the GUI helpers it was written for: nothing about reading "[0, 1, 2]" out
of a settings cell is a widget concern, and while it sat in the Tk helper
module every settings value that arrives as text depended on a GUI
toolkit being importable.
:param value: string representation of a list or tuple.
:returns: parsed list containing only ints, floats, or strings.
:raises ValueError: if the string is not a valid literal or contains
mixed / unsupported types.
"""
try:
parsed_value = ast.literal_eval(value)
if isinstance(parsed_value, list):
if all(isinstance(item, (int, float, str)) for item in parsed_value):
return parsed_value
raise ValueError("List contains mixed types or unsupported types")
if isinstance(parsed_value, tuple):
return list(parsed_value) if len(parsed_value) > 1 else [parsed_value[0]]
raise ValueError(f"Expected a list but got {type(parsed_value).__name__}")
except (ValueError, SyntaxError) as e:
raise ValueError(f"Invalid format for list: {value}. Error: {e}")
[docs]
def check_settings(vars_dict, expected_types, q=None):
"""Validate and coerce GUI-collected settings against expected types.
Iterates the widget map produced by the settings panel, parses each raw
string value into the type declared in ``expected_types`` (including
tuple-typed "or None" fields, lists, dicts and lists-of-lists), and
collects human-readable error messages instead of stopping at the first
failure. Errors are also forwarded to ``q`` for GUI display.
:param vars_dict: mapping ``key -> (label, widget, var, frame)`` from the settings panel.
:param expected_types: mapping ``key -> type`` (or tuple of accepted types).
:param q: optional queue used to surface error strings to the GUI. A private
Queue is created if None.
:returns: tuple ``(settings, errors)`` where ``settings`` is the parsed dict
and ``errors`` is the list of collected error messages.
"""
if q is None:
from multiprocessing import Queue
q = Queue()
settings = {}
errors = []
for key, (label, widget, var, _) in vars_dict.items():
if key not in expected_types and key not in category_keys:
errors.append(f"Warning: Key '{key}' not found in expected types.")
continue
value = var.get()
if value in ['None', '']:
value = None
expected_type = expected_types.get(key, str)
try:
if key in ["cell_plate_metadata", "timelapse_frame_limits", "png_size", "png_dims", "pathogen_plate_metadata", "treatment_plate_metadata", "timelapse_objects", "class_metadata", "crop_mode", "dialate_png_ratios"]:
if value is None:
settings[key] = None
continue
try:
parsed_value = ast.literal_eval(value)
except (ValueError, SyntaxError):
raise ValueError(f"Expected a list or list of lists but got an invalid format: {value}")
if isinstance(parsed_value, list):
if all(isinstance(i, list) for i in parsed_value) or all(not isinstance(i, list) for i in parsed_value):
settings[key] = parsed_value
else:
raise ValueError(f"Invalid format: '{key}' contains mixed types (single values and lists).")
else:
raise ValueError(f"Expected a list for '{key}', but got {type(parsed_value).__name__}.")
elif expected_type == list:
settings[key] = parse_list(value) if value else None
if isinstance(settings[key], list) and len(settings[key]) == 1:
settings[key] = settings[key][0]
elif expected_type == bool:
settings[key] = value.lower() in ['true', '1', 't', 'y', 'yes'] if isinstance(value, str) else bool(value)
elif expected_type == (int, type(None)):
if value is None or str(value).isdigit():
settings[key] = int(value) if value is not None else None
else:
raise ValueError(f"Expected an integer or None for '{key}', but got '{value}'.")
elif expected_type == (float, type(None)):
if value is None or (isinstance(value, str) and value.replace(".", "", 1).isdigit()):
settings[key] = float(value) if value is not None else None
else:
raise ValueError(f"Expected a float or None for '{key}', but got '{value}'.")
elif expected_type == (int, float):
try:
settings[key] = float(value) if '.' in str(value) else int(value)
except ValueError:
raise ValueError(f"Expected an integer or float for '{key}', but got '{value}'.")
elif expected_type == (bool, int):
if value is None:
settings[key] = None
else:
text = str(value).strip().lower()
if text in ('true', 't', 'y', 'yes'):
settings[key] = True
elif text in ('false', 'f', 'n', 'no'):
settings[key] = False
else:
try:
settings[key] = int(text)
except ValueError:
raise ValueError(
f"Expected True, False or an integer for '{key}', but got '{value}'.")
elif expected_type == (list, type(None)):
if value is None:
settings[key] = None
else:
try:
parsed_value = ast.literal_eval(value) if isinstance(value, str) else value
except (ValueError, SyntaxError):
raise ValueError(f"Expected a list or None for '{key}', but got: {value}")
if isinstance(parsed_value, tuple):
parsed_value = list(parsed_value)
if not isinstance(parsed_value, list):
raise ValueError(
f"Expected a list or None for '{key}', but got "
f"{type(parsed_value).__name__}.")
settings[key] = parsed_value
elif expected_type == (str, type(None)):
settings[key] = str(value) if value is not None else None
elif expected_type == (str, type(None), list):
if isinstance(value, list):
settings[key] = list(value) if value else None
elif isinstance(value, str):
settings[key] = str(value)
else:
settings[key] = None
elif expected_type == (str, bool):
if value is None or isinstance(value, bool):
settings[key] = value
elif str(value).strip().lower() in ("true", "false"):
settings[key] = str(value).strip().lower() == "true"
else:
settings[key] = str(value)
elif expected_type == dict:
try:
if isinstance(value, str):
parsed_dict = ast.literal_eval(value)
else:
raise ValueError("Expected a string representation of a dictionary.")
if not isinstance(parsed_dict, dict):
raise ValueError(f"Expected a dictionary for '{key}', but got {type(parsed_dict).__name__}.")
settings[key] = parsed_dict
except (ValueError, SyntaxError) as e:
settings[key] = {}
errors.append(f"Error: Invalid dictionary format for '{key}'. Expected type: dict. Error: {e}")
elif isinstance(expected_type, tuple):
for typ in expected_type:
try:
settings[key] = typ(value) if value else None
break
except (ValueError, TypeError):
continue
else:
raise ValueError(f"Value '{value}' for '{key}' does not match any expected types: {expected_type}.")
else:
try:
settings[key] = expected_type(value) if value else None
except (ValueError, TypeError):
raise ValueError(f"Expected type {expected_type.__name__} for '{key}', but got '{value}'.")
except (ValueError, SyntaxError) as e:
print(f"Processing key: '{key}' with value: '{value}' and expected type: {expected_type}")
expected_type_name = ' or '.join([t.__name__ for t in expected_type]) if isinstance(expected_type, tuple) else expected_type.__name__
errors.append(f"Error: '{key}' has invalid format. Expected type: {expected_type_name}. Got value: '{value}'. Error: {e}")
for error in errors:
q.put(error)
return settings, errors
[docs]
def set_annotate_default_settings(settings):
"""Populate default settings for the image annotation UI.
``crop_size`` was called ``img_size`` until 2026-09-19. A dict that
still says ``img_size`` has its value moved across before any default
lands, so an older settings file keeps the size it chose.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
_fold_renamed_settings(settings)
settings.setdefault('src', 'path')
settings.setdefault('image_type', 'cell_png')
settings.setdefault('channels', "r,g,b")
settings.setdefault('crop_size', 200)
settings.setdefault('annotation_column', 'test')
settings.setdefault('normalize_channels', None)
settings.setdefault('outline', None)
settings.setdefault('outline_threshold_factor', 1.25)
settings.setdefault('outline_sigma', 4)
settings.setdefault('edge_thickness', 0.1)
settings.setdefault('edge_transparency', 100)
settings.setdefault('edge_image', 'False')
settings.setdefault('object_size', (0,0))
settings.setdefault('percentiles', [2, 98])
settings.setdefault('measurement', '')
settings.setdefault('threshold', '')
settings.setdefault('threshold_direction', 'higher')
settings.setdefault('crop_source', 'png')
settings.setdefault('queue_by_uncertainty', False)
settings.setdefault('queue_measure', 'entropy')
settings.setdefault('queue_diversity', 'well')
settings.setdefault('queue_limit', 0)
settings.setdefault('cv_group_by', 'well')
return settings
[docs]
def set_default_generate_barecode_mapping(settings=None):
"""Return default settings for the barcode-mapping pipeline.
:param settings: optional dict to fill in place; a new dict is created if None.
:returns: the settings dict with defaults applied.
"""
if settings is None:
settings = {}
_fold_renamed_settings(settings)
settings.setdefault('src', 'path')
settings.setdefault('barcode_mismatches', 0)
settings.setdefault('regex', DEFAULT_BARCODE_REGEX)
settings.setdefault('target_sequence', 'TGCTGTTTCCAGCATAGCTCTTAAAC')
settings.setdefault('offset_start', -8)
settings.setdefault('window_length', 89)
settings.setdefault('column_csv', bundled_barcode_path('column'))
settings.setdefault('grna_csv', bundled_barcode_path('grna'))
settings.setdefault('row_csv', bundled_barcode_path('row'))
settings.setdefault('save_h5', True)
settings.setdefault('comp_type', 'zlib')
settings.setdefault('comp_level', 5)
settings.setdefault('chunk_size', 100000)
settings.setdefault('n_jobs', None)
settings.setdefault('ram_guard', True)
settings.setdefault('mode', 'paired')
settings.setdefault('single_direction', 'R1')
settings.setdefault('test', False)
settings.setdefault('fill_na', False)
return settings
[docs]
def get_default_generate_activation_map_settings(settings):
"""Populate default settings for generating model activation/CAM maps.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('dataset', 'path')
settings.setdefault('model_type', 'maxvit')
settings.setdefault('model_path', 'path')
settings.setdefault('image_size', 224)
settings.setdefault('batch_size', 64)
settings.setdefault('normalize', True)
settings.setdefault('cam_type', 'gradcam')
settings.setdefault('target_layer', None)
settings.setdefault('plot', False)
settings.setdefault('save', True)
settings.setdefault('normalize_input', True)
settings.setdefault('channels', [1,2,3])
settings.setdefault('overlay', True)
settings.setdefault('shuffle', True)
settings.setdefault('correlation', True)
settings.setdefault('manders_thresholds', [15,50, 75])
settings.setdefault('n_jobs', None)
settings.setdefault('ram_guard', True)
settings.setdefault('smoothgrad_samples', 0)
settings.setdefault('smoothgrad_sigma', 0.15)
settings.setdefault('occlusion_window', 8)
settings.setdefault('occlusion_stride', 4)
settings.setdefault('ig_steps', 50)
settings.setdefault('ig_baseline', 'zero')
settings.setdefault('attribution_steps', 12)
settings.setdefault('attribution_baseline', 'blur')
settings.setdefault('sanity_check', True)
settings.setdefault('object_type', 'cell')
settings.setdefault('counterfactuals', False)
settings.setdefault('counterfactual_crops', 256)
settings.setdefault('counterfactual_epochs', 30)
settings.setdefault('counterfactual_condition', 'class')
settings.setdefault('counterfactual_target', '')
settings.setdefault('counterfactual_generator', 'autoencoder')
return settings
[docs]
def get_analyze_plaque_settings(settings):
"""Populate default settings for plaque analysis.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src', 'path')
settings.setdefault('masks', True)
settings.setdefault('plaque_model', 'toxoplasma_plaque_v2')
settings.setdefault('plaque_mode', 'plaque')
settings.setdefault('figure_detector', 'toxoplasma_well_detector_v2')
settings.setdefault('figure_imgsz', '640,1280')
settings.setdefault('figure_confidence', 0.25)
settings.setdefault('figure_read_text', True)
settings.setdefault('confirm_annotations', False)
settings.setdefault('text_reach_above', 1.0)
settings.setdefault('text_reach_left', 1.0)
settings.setdefault('text_reach_below', 0.5)
settings.setdefault('text_use_above', True)
settings.setdefault('text_use_left', True)
settings.setdefault('text_use_below', True)
settings.setdefault('text_panel_reach', 1.0)
settings.setdefault('text_min_confidence', 0.0)
settings.setdefault('text_ignore', '')
settings.setdefault('text_order', 'above,left,below')
settings.setdefault('text_separator', ' / ')
settings.setdefault('text_reread', True)
settings.setdefault('text_reread_scale', 3)
settings.setdefault('well_detection', False)
settings.setdefault('well_confidence', 0.25)
settings.setdefault('well_pad', 0)
settings.setdefault('plate_format', None)
settings.setdefault('well_diameter_mm', None)
settings.setdefault('plaque_pixels_per_um', None)
settings.setdefault('plaque_formation_hours', None)
settings.setdefault('plaque_estimate_growth', False)
settings.setdefault('plaque_growth_reference_um', 893.8178699548309)
settings.setdefault('plaque_growth_reference_hours', 168.0)
settings.setdefault('colony_counting', False)
settings.setdefault('colony_dilution', 1)
settings.setdefault('colony_plated_volume_ul', 100)
settings.setdefault('colony_too_many', 300)
settings.setdefault('colony_too_few', 30)
settings.setdefault('colony_polarity', 'auto')
settings.setdefault('colony_threshold', 4.0)
settings.setdefault('colony_min_area_px', None)
settings.setdefault('colony_detector', None)
settings.setdefault('background', 200)
settings.setdefault('Signal_to_noise', 10)
settings.setdefault('CP_prob', 0)
settings.setdefault('diameter', 30)
settings.setdefault('batch_size', 50)
settings.setdefault('flow_threshold', 0.4)
settings.setdefault('save', True)
settings.setdefault('verbose', True)
settings.setdefault('resize', True)
settings.setdefault('target_height', 1120)
settings.setdefault('target_width', 1120)
settings.setdefault('rescale', False)
settings.setdefault('resample', False)
settings.setdefault('fill_in', True)
settings.setdefault('normalize', True)
settings.setdefault('channels', [0, 0])
settings.setdefault('percentiles', None)
settings.setdefault('invert', False)
settings.setdefault('grayscale', True)
settings.setdefault('remove_background', False)
settings.setdefault('model_name', 'cpsam')
return settings
[docs]
def set_graph_importance_defaults(settings):
"""Populate default settings for the "graph importance" plot utility.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('csvs','list of paths')
settings.setdefault('grouping_column','compartment')
settings.setdefault('data_column','compartment_importance_sum')
settings.setdefault(
'graph_type',
_graph_types.mark_to_start_on(
'categorical_continuous', 'jitter_box')[0])
settings.setdefault('save',False)
return settings
[docs]
def set_interpret_vision_model_defaults(settings):
"""Populate default settings for interpreting vision-model predictions.
Covers feature importance, permutation importance, and SHAP explanation
options over the cell/nucleus/pathogen/cytoplasm tables.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src','path')
settings.setdefault('scores','path')
settings.setdefault('tables',['cell', 'nucleus', 'pathogen','cytoplasm'])
settings.setdefault('feature_importance',True)
settings.setdefault('permutation_importance',False)
settings.setdefault('shap',True)
settings.setdefault('save',False)
settings.setdefault('nuclei_limit',1000)
settings.setdefault('pathogen_limit',1000)
settings.setdefault('top_features',30)
settings.setdefault('shap_sample',True)
settings.setdefault('n_jobs',-1)
settings.setdefault('ram_guard', True)
settings.setdefault('shap_approximate',True)
settings.setdefault('score_column','cv_predictions')
return settings
set_interperate_vision_model_defaults = set_interpret_vision_model_defaults
[docs]
def set_analyze_invasion_defaults(settings):
"""Populate default settings for the two-colour (red/green) invasion assay.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
_fold_renamed_settings(settings)
settings.setdefault('src','path')
settings.setdefault('parasite_table','pathogen')
settings.setdefault('compartment','pathogen')
settings.setdefault('outside_channel',1)
settings.setdefault('total_channel',0)
settings.setdefault('intensity_statistic','auto')
settings.setdefault('background_correction','none')
settings.setdefault('outside_threshold_method','otsu')
settings.setdefault('outside_threshold',None)
settings.setdefault('stain_baseline_wells', None)
settings.setdefault('control_quantile',0.99)
settings.setdefault('min_control_objects',10)
settings.setdefault('min_objects_for_threshold',10)
settings.setdefault('min_objects_for_bimodality',30)
settings.setdefault('bimodality_cutoff',0.5555555555555556)
settings.setdefault('threshold_agreement_tolerance',0.5)
settings.setdefault('threshold_sensitivity',0.25)
settings.setdefault('inflation_warn',0.05)
settings.setdefault('min_parasites_per_well',50)
settings.setdefault('min_parasite_area',0)
settings.setdefault('max_parasite_area',None)
settings.setdefault('min_total_intensity',None)
settings.setdefault('extracellular_class','attached')
settings.setdefault('seed_wells_from_cells',True)
settings.setdefault('cell_types',['HeLa'])
settings.setdefault('cell_plate_metadata',None)
settings.setdefault('pathogen_types',['pc'])
settings.setdefault('pathogen_plate_metadata',[['c1'], ['c2']])
settings.setdefault('treatments',None)
settings.setdefault('treatment_plate_metadata',None)
settings.setdefault('group_column','condition')
settings.setdefault('level','object')
settings.setdefault('change_plate',False)
settings.setdefault('qc_plot_max_panels',12)
settings.setdefault('cmap','viridis')
settings.setdefault('save',True)
settings.setdefault('verbose',False)
return settings
[docs]
def set_analyze_replication_defaults(settings):
"""Populate defaults for the parasites-per-vacuole replication assay.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('replication_method', 'direct_count')
for key, value in set_analyze_endodyogeny_defaults({}).items():
if key in ('tables', 'min_area_bin', 'max_area', 'max_bins', 'um_per_px',
'pathogen_limit', 'nuclei_limit', 'group_by_class', 'class_column'):
settings.setdefault(key, value)
settings.setdefault('src', 'path')
settings.setdefault('parasite_table', 'pathogen')
settings.setdefault('compartment', 'pathogen')
settings.setdefault('vacuole_key', 'auto')
settings.setdefault('vacuole_link_distance', None)
settings.setdefault('vacuole_link_factor', 1.5)
settings.setdefault('parasite_count_column', None)
settings.setdefault('min_parasite_area', 0)
settings.setdefault('max_parasite_area', None)
settings.setdefault('max_parasites_per_vacuole', 16)
settings.setdefault('require_host_cell', True)
settings.setdefault('seed_wells_from_cells', True)
settings.setdefault('non_power_of_two_warn', 0.2)
settings.setdefault('cell_types', ['HeLa'])
settings.setdefault('cell_plate_metadata', None)
settings.setdefault('pathogen_types', ['pc'])
settings.setdefault('pathogen_plate_metadata', [['c1'], ['c2']])
settings.setdefault('treatments', None)
settings.setdefault('treatment_plate_metadata', None)
settings.setdefault('group_column', 'condition')
settings.setdefault('level', 'object')
settings.setdefault('change_plate', False)
settings.setdefault('cmap', 'viridis')
settings.setdefault('save', True)
settings.setdefault('verbose', False)
return settings
[docs]
def set_analyze_endodyogeny_defaults(settings):
"""Populate default settings for endodyogeny (parasite division) analysis.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src','path')
settings.setdefault('tables',['cell', 'nucleus', 'pathogen', 'cytoplasm'])
settings.setdefault('cell_types',['Hela'])
settings.setdefault('cell_plate_metadata',None)
settings.setdefault('pathogen_types',['pc'])
settings.setdefault('pathogen_plate_metadata',[['c1'], ['c2']])
settings.setdefault('treatments',None)
settings.setdefault('treatment_plate_metadata',None)
settings.setdefault('min_area_bin',500)
settings.setdefault('max_area',1000000000)
settings.setdefault('group_column','condition')
settings.setdefault('compartment','pathogen')
settings.setdefault('pathogen_limit',1)
settings.setdefault('nuclei_limit',10)
settings.setdefault('level','object')
settings.setdefault('um_per_px',0.1)
settings.setdefault('max_bins',None)
settings.setdefault('save',False)
settings.setdefault('change_plate',False)
settings.setdefault('cmap','viridis')
settings.setdefault('verbose',False)
settings.setdefault('group_by_class',False)
settings.setdefault('class_column','predictions')
return settings
[docs]
def set_analyze_class_proportion_defaults(settings):
"""Populate default settings for class-proportion analysis across conditions.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src','path')
settings.setdefault('tables',['cell', 'nucleus', 'pathogen', 'cytoplasm'])
settings.setdefault('cell_types',['Hela'])
settings.setdefault('cell_plate_metadata',None)
settings.setdefault('pathogen_types',['nc','pc'])
settings.setdefault('pathogen_plate_metadata',[['c1'],['c2']])
settings.setdefault('treatments',None)
settings.setdefault('treatment_plate_metadata',None)
settings.setdefault('group_column','condition')
settings.setdefault('class_column','test')
settings.setdefault('pathogen_limit',1000)
settings.setdefault('nuclei_limit',1000)
settings.setdefault('level','well')
settings.setdefault('save',False)
settings.setdefault('verbose', False)
return settings
[docs]
def get_plot_data_from_csv_default_settings(settings):
"""Populate default settings for plotting data pulled from a CSV file.
:param settings: dict to fill in place.
:returns: the settings dict with defaults applied.
"""
settings.setdefault('src','path')
settings.setdefault('data_column','choose column')
settings.setdefault('grouping_column','choose column')
settings.setdefault(
'graph_type',
_graph_types.mark_to_start_on(
'categorical_continuous', 'jitter_box')[0])
settings.setdefault('save',False)
settings.setdefault('y_lim',None)
settings.setdefault('log_y',False)
settings.setdefault('log_x',False)
settings.setdefault('keep_groups',None)
settings.setdefault('representation','well')
settings.setdefault('theme','dark')
settings.setdefault('remove_outliers',False)
settings.setdefault('verbose',False)
return settings
[docs]
def get_automated_motility_assay_default_settings(settings):
"""Return default settings for the automated motility assay pipeline.
Combines array/filter parameters, XGBoost infection classifier settings,
and PCA/UMAP/t-SNE embedding options into a single settings dict.
:param settings: optional dict to fill in place; a new dict is created if None.
:returns: the settings dict with defaults applied.
"""
if settings is None:
settings = {}
_fold_renamed_settings(settings)
settings.setdefault('src', 'path')
settings.setdefault('channels', [0, 1, 2, 3])
settings.setdefault('cell_channel', 2)
settings.setdefault('nucleus_channel', 0)
settings.setdefault('pathogen_channel', 1)
settings.setdefault('tracked_object', 'cell')
settings.setdefault('reuse_existing_measurements', True)
settings.setdefault('infection_intensity_qc_scope', "per_well")
settings.setdefault('motility_analysis', False)
settings.setdefault('n_jobs', 8)
settings.setdefault('ram_guard', True)
settings.setdefault('max_displacement', 50.0)
settings.setdefault('track_outlier_zscore', 3.0)
settings.setdefault('drop_straight_tracks', False)
settings.setdefault('straightness_threshold', 0.95)
settings.setdefault('infection_intensity_strategy', 'xgboost')
settings.setdefault('infection_intensity_mode', "relabel")
settings.setdefault('db_table_name', "timelapse_object_measurements")
settings.setdefault('infection_intensity_n_bins', 64)
settings.setdefault('infection_intensity_qc_graphs', True)
settings.setdefault('pixels_per_um', 1.78)
settings.setdefault('seconds_per_frame', 60)
settings.setdefault('motility_xlim', (100, -100))
settings.setdefault('motility_ylim', (100, -100))
settings.setdefault('infection_xgb_n_estimators', 200)
settings.setdefault('infection_xgb_max_depth', 3)
settings.setdefault('infection_xgb_learning_rate', 0.1)
settings.setdefault('infection_xgb_subsample', 0.8)
settings.setdefault('infection_xgb_colsample_bytree', 0.8)
settings.setdefault('infection_xgb_reg_lambda', 1.0)
settings.setdefault('infection_xgb_random_state', 42)
settings.setdefault('infection_xgb_n_jobs', -1)
settings.setdefault('infection_xgb_proba_threshold', 0.5)
settings.setdefault('infection_xgb_margin', 0.15)
settings.setdefault('infection_xgb_top_features', 20)
settings.setdefault('infection_xgb_proba_column', 'infection_xgb_proba')
settings.setdefault('infection_xgb_drop_ambiguous', True)
settings.setdefault('infection_xgb_ambiguous_low', 0.25)
settings.setdefault('infection_xgb_ambiguous_high', 0.75)
settings.setdefault('infection_xgb_min_cells_per_class', 10)
settings.setdefault('infection_pca_random_state', 42)
settings.setdefault('infection_pca_pathogen_weight', 2.0)
settings.setdefault('infection_pca_log_intensity', False)
settings.setdefault('infection_pca_max_cells', 50000)
settings.setdefault('infection_pca_min_gt_separation', 0.2)
settings.setdefault('infection_pca_min_silhouette', 0.05)
settings.setdefault('infection_pca_umap_search', True)
settings.setdefault('infection_pca_umap_n_neighbors_grid', [5, 10, 15, 30])
settings.setdefault('infection_pca_umap_min_dist_grid', [0.0, 0.05, 0.1, 0.3])
settings.setdefault('infection_pca_umap_n_neighbors', 15)
settings.setdefault('infection_pca_umap_min_dist', 0.1)
settings.setdefault('infection_pca_tsne_search', True)
settings.setdefault('infection_pca_tsne_perplexity_grid', [15.0, 30.0, 45.0])
settings.setdefault('infection_pca_tsne_learning_rate_grid', [200.0, 500.0])
settings.setdefault('infection_pca_tsne_perplexity', 30.0)
return settings
#: Measurement families whose meaning depends on there being MANY separable
#: objects of the kind inside one cell.
COUNT_DEPENDENT_MEASUREMENTS = ('spatial_measurements', 'object_distances')
#: How many objects of each type a cell holds, which is what decides whether
#: a count-dependent measurement means anything.
#:
#: ``'many'`` -- separable objects, the counting families are the phenotype.
#: ``'one'`` -- one connected structure per cell, so a count is a
#: segmentation artefact rather than a measurement.
#: ``'depends'`` -- the category holds both kinds, so what decides it is
#: the structure in front of you rather than the name.
#: ``'unknown'`` -- 'custom' recommends nothing and is not told what it is.
ORGANELLE_OBJECTS_PER_CELL = {
'custom': 'unknown',
'punctate': 'many',
'vesicular': 'many',
'crescent': 'many',
'spherical': 'depends',
'toroidal': 'depends',
'filamentous': 'one',
'tubular': 'one',
'reticular': 'one',
'cisternal': 'one',
}
#: Why, in the user's own vocabulary. Shown beside the caveat rather than
#: left for the reader to work out, because "meaningless" without a reason
#: reads as a bug in the tool.
ORGANELLE_COUNT_REASONS = {
'filamentous': (
"microtubules and actin filaments are one connected cytoskeleton "
"per cell, so a neighbour count measures where the segmentation "
"broke the network rather than anything about the cell"),
'tubular': (
"smooth ER and a healthy mitochondrial network are continuous "
"through the cell, so counting separate objects counts breaks in "
"the tubule, not organelles"),
'reticular': (
"a reticulum is one connected mesh filling the cell: its neighbour "
"count is zero and its nearest-neighbour distance is undefined"),
'cisternal': (
"a Golgi stack or a nuclear envelope is one structure per cell, so "
"the object count is set by how the sheet happened to be cut"),
'spherical': (
"this category spans one-per-cell structures (nucleus, nucleolus) "
"and many-per-cell ones (condensates, swollen mitochondria), so "
"whether counting means anything depends on which you imaged"),
'toroidal': (
"midbodies are one per dividing cell while stressed mitochondria "
"and nuclear pores are many, so whether counting means anything "
"depends on which you imaged"),
}
[docs]
def organelle_counting_is_meaningful(organelle_type):
"""Whether count-dependent measurements mean anything for a type.
:param organelle_type: a key of
:data:`spacr.organelle_types.ORGANELLE_TYPES`.
:returns: True for a type that is many separable objects per cell,
False for one connected structure, and True for 'depends' and
'unknown' -- the measurement is still produced and the caveat is
what carries the doubt. Refusing to measure on a maybe would delete
a real number for half the structures in the category.
"""
return ORGANELLE_OBJECTS_PER_CELL.get(
str(organelle_type or DEFAULT_ORGANELLE_TYPE), 'unknown') != 'one'
[docs]
def organelle_measurement_caveats(settings):
"""What a measure run should say about its own organelle numbers.
One entry per enabled organelle slot whose type makes a count-dependent
family read differently from the way it reads for cells: ``(slot label,
setting name, reason)``.
A slot is ENABLED when it has a mask dimension to measure. A slot with
no type set says nothing, because a settings file written before the
type existed is not making a claim about what it imaged.
:param settings: a measure settings dict.
:returns: a list of ``(label, setting, reason)``, empty when every
enabled slot is a many-per-cell type or has no type at all.
"""
out = []
for role in active_organelle_roles(settings):
mask_dim = settings.get(f'{role}_mask_dim')
channel = settings.get(f'{role}_channel')
if mask_dim is None and channel is None:
continue
kind = settings.get(f'{role}_type')
if not kind:
continue
kind = str(kind)
verdict = ORGANELLE_OBJECTS_PER_CELL.get(kind, 'unknown')
if verdict in ('many', 'unknown'):
continue
reason = ORGANELLE_COUNT_REASONS.get(kind, '')
label = organelle_slot_label(role)
for setting in COUNT_DEPENDENT_MEASUREMENTS:
if not settings.get(setting):
continue
qualifier = 'does not mean' if verdict == 'one' else 'may not mean'
out.append((
label, setting,
f"{qualifier} for a {kind} organelle what it means for a "
f"cell: {reason}"))
return out
[docs]
def explain_organelle_measurements(settings):
"""Print the caveats above, in the voice the type preset already uses.
Silent when there is nothing to say, so a run measuring punctate
organelles is not given a paragraph telling it everything is fine.
:param settings: a measure settings dict.
:returns: the caveats, so a caller can show them somewhere other than
the console.
"""
caveats = organelle_measurement_caveats(settings)
for label, setting, reason in caveats:
print(f"[organelle] {label}: {setting} {reason}.")
return caveats
def _set_organelle_defaults(settings):
"""Fill in default values for the organelle_* keys of every slot.
HOW MANY SLOTS is `number_of_organelles`, and the keys of each are
generated from it rather than written out. Slots the count no longer
reaches keep whatever they hold: the loop below runs over
:func:`spacr.organelle_types.declared_organelle_roles`, which is the
active slots plus any further slot this dict already carries a key for,
and every write is a ``setdefault``. Lowering the number is therefore
reversible -- raising it again finds the old answers still there.
An old settings file without the per-slot switch keeps the old shared
behaviour: the slot's channel follows the generic remove_background.
"""
settings.setdefault(NUMBER_OF_ORGANELLES, organelle_count(settings))
defaults = {
'organelle_channel': None,
'organelle_type': DEFAULT_ORGANELLE_TYPE,
'organelle_morphology': 'spots',
'organelle_method': 'otsu',
'organelle_diameter': 30,
'organelle_model_name': 'cpsam',
'organelle_min_area': 10,
'organelle_max_area': None,
'organelle_min_intensity': 0.0,
'organelle_max_intensity': 0.0,
'organelle_remove_border': False,
'organelle_background': 100,
'organelle_signal_to_noise': 10,
'organelle_rolling_ball': False,
'organelle_rolling_ball_radius': 50,
'organelle_clahe': False,
'organelle_clahe_clip_limit': 0.01,
'organelle_mask_within_cells': False,
'organelle_log_min_sigma': 1,
'organelle_log_max_sigma': 10,
'organelle_log_num_sigma': 10,
'organelle_log_threshold': 0.01,
'organelle_dog_sigma_low': 1.0,
'organelle_dog_sigma_high': 3.0,
'organelle_tophat_radius': 5,
'organelle_watershed_spots': True,
'organelle_ridge_sigmas': [1, 2, 3],
'organelle_ridge_filter': 'frangi',
'organelle_skeletonize': False,
'organelle_network_threshold': 'otsu',
'organelle_hysteresis_low': 0.2,
'organelle_hysteresis_high': 0.6,
'organelle_unet_model_path': None,
'organelle_unet_threshold': 0.5,
'organelle_adaptive_block_size': 51,
'organelle_adaptive_offset': 5,
'organelle_morph_radius': 3,
'organelle_fill_holes': 64,
'organelle_ring_sigma_inner': 1.0,
'organelle_ring_sigma_outer': 3.0,
'organelle_ring_min_prominence': 0.1,
'organelle_ring_fill_method': 'flood',
'organelle_cellprob_threshold': 0.0,
'organelle_flow_threshold': 0.4,
'organelle_resample': True,
}
settings.update(apply_preset(settings,
explain=bool(settings.get('verbose'))))
for key, val in defaults.items():
settings.setdefault(key, val)
for role in declared_organelle_roles(settings)[1:]:
view = {'verbose': settings.get('verbose', False)}
prefix = f'{role}_'
for key, value in settings.items():
if str(key).startswith(prefix):
view[f"organelle_{str(key)[len(prefix):]}"] = value
view = apply_preset(view, explain=bool(settings.get('verbose')))
for key, value in defaults.items():
slot_key = _organelle_slot_key(key, role)
base_value = view.get(key, value)
settings.setdefault(slot_key, deepcopy(base_value))
settings.setdefault(_background_switch_key(role),
settings.get('remove_background', False))
return settings
from . import illumination as _illumination # noqa: E402,F401
from . import ops_settings as _ops_settings # noqa: E402,F401
expected_types.update({
'hp_vacuole_table': str, 'hp_vacuole_prefix': str,
'hp_reference_table': str, 'hp_reference_prefix': str,
'hp_marker_channels': list, 'hp_marker_thresholds': dict,
'hp_parasite_table': str, 'hp_parasite_parent': str, 'hp_count_column': str,
})
tooltips.update({
'hp_vacuole_table': '(str) - Measurement table with one object per whole vacuole and a cell_id link to its host. Use whole-vacuole masks, not individual parasite masks. Default pathogen. API: spacr.host_pathogen.analyze_host_pathogen.',
'hp_vacuole_prefix': '(str) - Select the measured intensity column family used as the numerator of each recruitment ratio, for example pathogen_channel_0_mean_intensity. Match the prefix in the selected vacuole table; changing it does not change masks or infer a different object type. Default pathogen. API: spacr.host_pathogen.summarize_tables.',
'hp_reference_table': '(str) - Select the per-host compartment used for recruitment-ratio denominators. Its object_label must identify the host cell so each vacuole is compared with its own host reference. Missing or invalid reference intensities produce unknown marker calls rather than negative calls. Default cytoplasm. API: spacr.host_pathogen.summarize_tables.',
'hp_reference_prefix': '(str) - Prefix of reference-channel mean intensities. Missing or nonpositive reference values yield unknown ratios, never infinity or negative calls. Default cytoplasm. API: spacr.host_pathogen.summarize_tables.',
'hp_marker_channels': '(list) - Zero-based intensity channels whose vacuole-to-host-reference ratios are reported. Each selected channel needs matching intensity columns in both compartments; its threshold is configured separately. Adding a channel adds a marker result without changing host infection denominators. Default [0]. API: spacr.host_pathogen.summarize_tables.',
'hp_marker_thresholds': '(dict) - Channel-to-ratio cutoffs, for example {0: 2.0, 1: 1.5}; ratios at or above the cutoff are positive. Calibrate cutoffs with assay controls. Unspecified channels remain unclassified. Default {}. API: spacr.host_pathogen.summarize_tables.',
'hp_parasite_table': '(str) - Optional table containing one row per segmented parasite and an explicit parent-vacuole column. Leave empty if parasite counts are unavailable. Do not also set hp_count_column. Default empty. API: spacr.host_pathogen.summarize_tables.',
'hp_parasite_parent': '(str) - Parent-vacuole label column in the selected parasite table. Host cell IDs cannot substitute for vacuole IDs. Unmatched parasites are exported separately. Default pathogen_id. API: spacr.host_pathogen.summarize_tables.',
'hp_count_column': '(str) - Optional measured count column on each vacuole. Nonnegative integer counts are accepted; missing values remain unknown. Alternative to a linked parasite table. Default empty. API: spacr.host_pathogen.summarize_tables.',
})
ALPHA_KINDS = ('settings', 'choices', 'widgets', 'apps', 'models',
'module_settings')
ALPHA_FEATURES = {
426: {
'settings': ('timeflows_model',),
'choices': {'timelapse_mode': ('timeflows',)},
},
493: {
'settings': ('mask_parallel', 'mask_gpu_indices'),
'widgets': ('DistributedAllocatedGpus', 'MaskGpuProgress'),
},
535: {
'settings': ('cell_cycle', 'cell_cycle_method', 'cell_cycle_channel',
'cell_cycle_gates', 'cell_cycle_mitotic_ratio',
'cell_cycle_fucci_channels', 'cell_cycle_labels',
'cell_cycle_model', 'cell_cycle_epochs'),
},
539: {
'settings': ('bleach_correction',),
},
566: {
'settings': ('measure_gpu',),
},
576: {
'settings': ('measurement_backend', 'measurement_backend_target'),
'widgets': ('ALPHA_FEATURES576', 'MeasurementMigrationAction'),
},
540: {
'settings': ('viability', 'viability_dead_channel',
'viability_live_channel', 'viability_thresholds',
'viability_negative_wells', 'viability_positive_wells',
'viability_plate_map'),
},
541: {
'settings': ('confluency', 'confluency_source', 'confluency_channel',
'confluency_window', 'confluency_qc_threshold'),
'widgets': ('MeasureConfluencyToggle',),
},
542: {
'settings': ('colony_counting', 'colony_dilution',
'colony_plated_volume_ul', 'colony_too_many',
'colony_too_few', 'colony_polarity', 'colony_threshold',
'colony_min_area_px', 'colony_detector'),
'models': ('colony_yolo11n_makrai_v1',),
},
536: {
'settings': ('wound_closure', 'wound_source', 'wound_channel',
'wound_window', 'wound_threshold',
'wound_hours_per_frame', 'wound_conditions'),
'widgets': ('MeasureWoundToggle',),
},
544: {
'widgets': ('AnnotateBlindToggle', 'MakeMasksBlindToggle'),
},
545: {
'widgets': ('MakeMasksRoisButton',),
},
547: {
'settings': ('profiling', 'profiling_metadata',
'profiling_treatment_column',
'profiling_negative_control', 'profiling_normalization',
'profiling_feature_selection',
'profiling_correlation_threshold',
'profiling_phenotype_column', 'profiling_databases'),
},
548: {
'settings': ('watch_folder', 'watch_pipeline', 'watch_normalization_pool', 'watch_measure_settings',
'watch_classify_settings',
'watch_settle_seconds', 'watch_poll_seconds',
'watch_idle_minutes'),
'widgets': ('WatchFolderProgress', 'WatchLivePlate'),
},
549: {
'settings': ('microscope_feedback', 'microscope_driver',
'microscope_simulated_folder', 'microscope_positions',
'microscope_stage_transform', 'microscope_event_table',
'microscope_event_query', 'microscope_max_events',
'microscope_timepoints', 'microscope_interval_seconds'),
},
550: {
'settings': ('cloud_anonymous', 'cloud_profile', 'cloud_endpoint',
'cloud_cache', 'cloud_wells', 'cloud_fields',
'cloud_level', 'cloud_results'),
'widgets': ('CloudSourceBrowse',),
},
546: {
'settings': ('cellprofiler_pipeline',),
'models': ('cellprofiler_v1',),
},
551: {
'models': ('stardist_v1', 'stardist_2D_versatile_fluo',
'stardist_2D_versatile_he', 'stardist_2D_paper_dsb2018'),
},
552: {
'models': ('instanseg_v1', 'instanseg_fluorescence_nuclei_and_cells',
'instanseg_brightfield_nuclei'),
},
553: {
'models': ('omnipose_v1', 'omnipose_bact_phase_omni',
'omnipose_bact_fluor_omni', 'omnipose_worm_omni',
'omnipose_worm_bact_omni', 'omnipose_worm_high_res_omni',
'omnipose_cyto2_omni'),
},
554: {
'choices': {'ops_spot_detector': ('spotiflow',)},
'models': ('spotiflow_v1', 'spotiflow_general', 'spotiflow_hybiss',
'spotiflow_synth_complex', 'spotiflow_fluo_live'),
},
555: {
'widgets': ('MakeMasksPromptCategory',),
'models': ('microsam_v1',),
},
556: {
'choices': {'timelapse_mode': ('sam2',)},
'models': ('sam2_v1',),
'widgets': ('MakeMasksSam2Button',),
},
565: {
'widgets': ('AnnotateFindSimilar', 'AnnotateSimilarityOptions',
'AnnotateSimilarAddPlate', 'AnnotateSimilarClearPlates',
'AnnotateSimilarResultPlate', 'AnnotateSimilarFeatureKind',
'EmbeddingsSaveForSimilarity'),
},
560: {
'widgets': ('EmbeddingsFoundationLabel', 'EmbeddingsFoundationPicker',
'EmbeddingsSubCellChannelsButton',
'EmbeddingsSubCellChannelsDialog',
'EmbeddingsSubCellMicrotubulesChannel',
'EmbeddingsSubCellErChannel',
'EmbeddingsSubCellDnaChannel',
'EmbeddingsSubCellProteinChannel',
'EmbeddingsSubCellChannelsProblem',
'EmbeddingsCellDinoConfigButton',
'EmbeddingsCellDinoDialog',
'EmbeddingsCellDinoFactory',
'EmbeddingsCellDinoCheckpointPath',
'EmbeddingsCellDinoBrowse',
'EmbeddingsCellDinoDigest',
'EmbeddingsCellDinoPlane1',
'EmbeddingsCellDinoPlane2',
'EmbeddingsCellDinoPlane3',
'EmbeddingsCellDinoPlane4',
'EmbeddingsCellDinoPlane5',
'EmbeddingsCellDinoProblem',
'EmbeddingsLabelsButton', 'EmbeddingsUseForLabel',
'EmbeddingsUseForPicker', 'EmbeddingsUseForRun'),
},
562: {
'widgets': ('EmbeddingsWellMilButton',),
},
561: {
'widgets': ('EmbeddingsDinoPretrainButton',),
},
558: {
'widgets': ('CellposeWorkbenchVirtualStain', 'MaskVirtualStainApply',
'MakeMasksVirtualStainApply'),
},
685: {
'widgets': ('MakeMasksVvvvExport', 'MakeMasksVvvvExportOnSave'),
},
570: {
'widgets': ('ControlChartHitPanel', 'ControlChartHitsSection',
'ControlChartExportHits'),
},
571: {
'settings': ('time_to_event', 'time_to_event_object',
'time_to_event_mode', 'time_to_event_column',
'time_to_event_threshold', 'time_to_event_persist',
'time_to_event_origin', 'time_to_event_min_frames',
'time_to_event_hours_per_frame', 'time_to_event_group',
'time_to_event_conditions', 'time_to_event_reference',
'time_to_event_covariates'),
},
572: {
'widgets': ('FigureIntegrityCheck',),
},
573: {
'widgets': ('AnalysisLockButton',),
},
580: {
'settings': ('intensity_calibration', 'intensity_calibration_wells',
'intensity_calibration_statistic',
'intensity_calibration_offset'),
},
577: {
'widgets': ('NotifyTabHelp', 'NotifyRunsEnabled', 'NotifyRunsWhen',
'NotifyRunsMinMinutes', 'NotifyDesktop', 'NotifyEmail',
'NotifySmtpHost', 'NotifySmtpPort', 'NotifySmtpSecurity',
'NotifySmtpUser', 'NotifySmtpPassword', 'NotifyEmailFrom',
'NotifyEmailTo', 'NotifySlack', 'NotifySlackWebhook',
'NotifyNtfy', 'NotifyNtfyServer', 'NotifyNtfyTopic',
'NotifyNtfyToken', 'NotifySendTest', 'NotifyForgetSecrets',
'NotifyTestResult', 'NotifyTeams', 'NotifyTeamsWebhook',
'NotifyWebhook', 'NotifyWebhookUrl', 'NotifyWebhookToken'),
},
581: {
'settings': ('anndata_format', 'anndata_tidy_dir'),
},
538: {
'settings': ('unmix', 'unmix_controls', 'unmix_background_percentile'),
},
557: {
'settings': ('n2v_denoise', 'n2v_model', 'n2v_epochs'),
'models': ('careamics_v1',),
},
574: {
'widgets': ('ReportArchivePackage', 'ArchiveScreenSources'),
},
579: {
'widgets': ('ReportZenodoDeposit',),
},
575: {
'widgets': ('RunHistoryExportWorkflow',),
},
537: {
'settings': ('timelapse_lineage', 'timelapse_lineage_color_by',
'timelapse_lineage_max_distance',
'timelapse_lineage_min_division_h'),
},
567: {
'models': ('videomae_v1',),
'settings': ('timelapse_events', 'timelapse_events_annotations',
'timelapse_events_model', 'timelapse_events_window',
'timelapse_events_threshold',
'timelapse_events_conditions', 'timelapse_events_encoder',
'timelapse_events_video_checkpoint', 'timelapse_events_video_channels',
'timelapse_events_video_device'),
'widgets': ('TimelapseEventAnnotationButton',
'TimelapseEventAnnotationDialog',
'TimelapseEventTracksPath', 'TimelapseEventSequencePath',
'TimelapseEventOutputPath', 'TimelapseEventOpenField',
'TimelapseEventConfirmField', 'TimelapseEventFramePreview',
'TimelapseEventTrack', 'TimelapseEventFrame',
'TimelapseEventChannel', 'TimelapseEventName',
'TimelapseEventRows', 'TimelapseEventAdd',
'TimelapseEventRemove', 'TimelapseEventSave',
'TimelapseEventBrowseTracks',
'TimelapseEventBrowseSequence',
'TimelapseEventBrowseFolder',
'TimelapseEventBrowseOutput', 'TimelapseEventClose'),
},
583: {
'settings': ('plate_barcode_source', 'plate_barcodes',
'plate_barcode_column', 'plate_barcode_token_env'),
'widgets': ('ConvertPlateBarcodeLinkage',),
},
543: {
'settings': ('illumination_vendor_profile', 'illumination_vendor_channel_map'),
},
584: {
'widgets': ('ControlChartChemistry', 'ControlChartChemistrySection'),
},
585: {
'widgets': ('PowerArrayedPlanner',),
}, 563: {
'widgets': ('ControlChartAnomaly', 'ControlChartAnomalySection'),
},
568: {
'widgets': ('MakeMasksUncertaintyButton', 'MakeMasksUncertaintySetting',
'MakeMasksUncertaintyEnsembleSetting'),
},
470: {
'settings': ('real_object_classifier', 'real_object_threshold'),
},
578: {
'settings': ('robustness_report', 'robustness_fields',
'robustness_crop', 'robustness_diameter_factors',
'robustness_flow_thresholds',
'robustness_cellprob_thresholds',
'robustness_enhancement', 'robustness_tolerance'),
},
559: {
'settings': ('image_qc_classifier', 'image_qc_classifier_model',
'image_qc_classifier_labels',
'image_qc_classifier_threshold'),
'widgets': ('AnnotateFieldQCButton', 'AnnotateFieldQCDialog',
'QCClassifierCard'),
},
582: {
'widgets': ('PluginCatalogueHelp', 'PluginCatalogueSource',
'PluginCatalogueLoad', 'PluginCatalogueTable',
'PluginCatalogueInstall', 'PluginCatalogueUninstall',
'PluginCatalogueOpen', 'PluginCatalogueStatus'),
},
534: {
'widgets': ('MapBarcodesSpatialToggle', 'MapBarcodesSpatialCard'),
},
591: {
'module_settings': {
app_key: (
'illumination_correction', 'illumination_model',
'illumination_estimator', 'illumination_degree',
'illumination_dark', 'illumination_per_plate',
'illumination_max_fields', 'illumination_qc',
'illumination_on_missing',
'psf_measurement_source', 'psf_operation', 'psf_source',
'psf_objective', 'psf_path', 'psf_image_sampling_um',
'psf_kernel_sampling_um', 'psf_fwhm_um', 'psf_iterations',
'enhance_background', 'enhance_background_radius',
'enhance_background_scale', 'enhance_denoise',
'enhance_denoise_strength', 'enhance_percentile_clip',
'enhance_percentile_low', 'enhance_percentile_high',
'enhance_gamma', 'enhance_log', 'enhance_log_gain',
'enhance_sqrt', 'enhance_clahe', 'enhance_clahe_tile',
'enhance_clahe_clip', 'enhance_equalize', 'enhance_sharpen',
'enhance_sharpen_radius', 'enhance_sharpen_amount',
)
for app_key in ('mask', 'timelapse')
},
},
404: {
'choices': {'segmentation_backend': ('dinocell',)},
'models': ('dinocell_v1',),
},
405: {
'choices': {'segmentation_backend': ('samcell',)},
'models': ('samcell_v1',),
},
475: {
'choices': {'ops_spot_detector': ('spotnet',)},
'models': ('spotnet_v1',),
},
501: {
'settings': ('plaque_estimate_growth', 'plaque_growth_reference_um',
'plaque_growth_reference_hours'),
'widgets': ('PlaqueEstimateScaleTime', 'PlaqueEstimateScaleTimeNote'),
},
564: {
'settings': ('counterfactuals', 'counterfactual_crops',
'counterfactual_epochs', 'counterfactual_condition',
'counterfactual_target', 'counterfactual_generator'),
'widgets': ('ActivationCounterfactualViewer',),
},
508: {
'widgets': ('MakeMasksUseInMaskGeneration',),
},
633: {
'widgets': ('TrellisTestDataButton', 'FeatureExplorerTestDataButton',
'OutliersTestDataButton', 'ControlChartTestDataButton',
'DataManagerTestDataButton', 'EmbeddingsTestDataButton',
'PowerTestDataButton', 'ConvertTestDataButton',
'LayerViewerTestDataButton', 'ExternalMasksTestDataButton',
'PipelineGraphTestDataButton',
'ProjectBrowserTestDataButton',
'DoseResponseTestDataButton', 'ProfilerTestDataButton',
'RunCompareTestDataButton', 'RunHistoryTestDataButton',
'TrainCompareTestDataButton'),
},
}
ALPHA_SPECIES = {
634: {
'apps': ('plasmodium', 'candida', 'trypanosoma', 'leishmania',
'giardia', 'virus', 'mammalian'),
'widgets': ('PlasmodiumOrganismPage', 'CandidaOrganismPage',
'TrypanosomaOrganismPage', 'LeishmaniaOrganismPage',
'GiardiaOrganismPage', 'VirusOrganismPage',
'MammalianOrganismPage'),
},
}
_SPECIES_WITH_PUBLISHED_LESSONS = frozenset()
"""Alpha species whose pages still have published lessons.
Tutorial navigation and the module workflow map keep these pages in place
until their lessons are retired; the app itself hides them like the rest of
``ALPHA_SPECIES``. Empty since tutorial wave 3 (2026-10-04) withdrew
83_plasmodium and 84_candida.
"""
def _alpha_names(kind, app_key=None):
"""Every name registered as alpha under ``kind``, across all items.
``ALPHA_FEATURES`` is the one registry of everything built from
``features/future`` that ships as an alpha feature (see
``features/README.md``): hidden unless Preferences -> Show alpha
features is on. Each entry is keyed by the item
number and lists what that item adds under the kinds in ``ALPHA_KINDS``:
``settings`` (settings keys, hidden from the form, the settings search
and its counts), ``choices`` (``{key: (dropdown values,)}``), ``widgets``
(Qt object names of buttons, checkboxes, labels, menu actions or
panels), ``apps`` (module keys: tile, sidebar, menu and palette together)
``models`` (Model Zoo keys, names or family stems) and
``module_settings`` (``{module key: (settings keys,)}``, settings that are
alpha on those modules' forms only, where the same key is an ordinary
setting of another module). A feature is marked in this one place and
promoted out of alpha by deleting its entry. Organism pages are listed
in ``ALPHA_SPECIES`` instead and are included here too.
Hiding is a display decision only: a saved or typed alpha setting still
reaches the run, and headless and command-line runs never consult the
registry. The one exception is run-finished notifications, which are
configured only in Preferences and are sent, from the app or the command
line, only while the gate shows them.
:param kind: one of ``ALPHA_KINDS``.
:param app_key: with ``kind='settings'``, the module whose form is asked
about, so its ``module_settings`` are included; without it only the
settings alpha on every module.
:returns: a frozenset of names; for ``choices`` the settings keys that
carry alpha entries, for ``module_settings`` every settings key alpha
on some module.
:raises ValueError: for a kind that is not in ``ALPHA_KINDS``.
"""
if kind not in ALPHA_KINDS:
raise ValueError(
f'unknown alpha kind {kind!r}; expected one of {ALPHA_KINDS}')
names = set()
for entry in (*ALPHA_FEATURES.values(), *ALPHA_SPECIES.values()):
if kind == 'module_settings':
for keys in (entry.get(kind) or {}).values():
names.update(keys or ())
continue
names.update(entry.get(kind, ()) or ())
if kind == 'settings' and app_key is not None:
scoped = entry.get('module_settings') or {}
names.update(scoped.get(str(app_key), ()) or ())
return frozenset(str(name) for name in names)
def _alpha_species_names(kind):
"""The names of ``kind`` registered in ``ALPHA_SPECIES`` only.
``ALPHA_SPECIES`` lists the alpha organism pages. It has the shape of
``ALPHA_FEATURES`` and is counted with it by :func:`_alpha_names`, but
Preferences -> Show alpha species shows or hides it instead of Show
alpha features, so the two switches are independent. Toxoplasma is not
listed and is always shown.
:param kind: one of ``ALPHA_KINDS``.
:returns: a frozenset of names that Show alpha species gates.
:raises ValueError: for a kind that is not in ``ALPHA_KINDS``.
"""
if kind not in ALPHA_KINDS:
raise ValueError(
f'unknown alpha kind {kind!r}; expected one of {ALPHA_KINDS}')
names = set()
for entry in ALPHA_SPECIES.values():
names.update(entry.get(kind, ()) or ())
return frozenset(str(name) for name in names)
def _alpha_choices(key):
"""The dropdown entries of settings ``key`` that are alpha.
:param key: a settings key, such as ``'timelapse_mode'``.
:returns: a frozenset of the entries' values, empty when none are alpha.
"""
values = set()
for entry in (*ALPHA_FEATURES.values(), *ALPHA_SPECIES.values()):
values.update((entry.get('choices') or {}).get(str(key), ()) or ())
return frozenset(str(value) for value in values)
def _is_alpha(kind, name, choice=None):
"""Whether ``name`` of ``kind`` is registered with the alpha gate.
:param kind: one of ``ALPHA_KINDS``.
:param name: the settings key, object name, module key or model key.
:param choice: with ``kind='choices'``, the dropdown entry asked about;
without it the question is whether ``name`` has any alpha entry.
:returns: True when registered.
"""
if kind == 'choices' and choice is not None:
return str(choice) in _alpha_choices(name)
return str(name) in _alpha_names(kind)