Source code for spacr.qt.regex_detect

"""
Filename-metadata regex helpers for drag-and-drop and manual configuration.

Central place for four related concerns:

* :func:`apply_regex` — run a compiled regex over a list of filenames,
  return a list of :class:`MetadataRecord` dicts (one per file that
  matched).
* :func:`validate_records` — check whether the parsed records supply
  the fields the rest of spaCR needs (``wellID``/``fieldID`` +
  ``chanID`` for multi-channel; ``fieldID`` for single-channel).
  Returns human-friendly warning strings for anything missing.
* :func:`auto_detect_regex` — heuristic detector that tries each
  built-in regex, and if none fit, synthesises a fresh one from the
  common shape of the sampled filenames.
* :func:`tabulate_records` — render a small aligned text table of
  records for the Console.

Public constants:

* :data:`BUILTIN_REGEXES` — ordered ``{label: pattern}`` map tried by
  :func:`auto_detect_regex` before falling back to synthesis.
* :data:`REQUIRED_MULTICHANNEL` / :data:`REQUIRED_SINGLECHANNEL` —
  which group names must be present for spaCR to be happy.
"""
from __future__ import annotations

import random
import re
from dataclasses import dataclass
from typing import Dict, List, Optional, Sequence, Set, Tuple


CELLVOYAGER = (
    r"(?P<plateID>.*)_(?P<wellID>.*)_T(?P<timeID>.*)F(?P<fieldID>.*)"
    r"L(?P<laserID>..)A(?P<AID>..)Z(?P<sliceID>.*)C(?P<chanID>.*)"
    r"\.(?:tif|tiff|png|jpg|jpeg)$"
)

YOKOGAWA = (
    r"(?P<plateID>.*)_(?P<wellID>[A-Z]\d{2})_"
    r"T(?P<timeID>\d{4})F(?P<fieldID>\d{3})"
    r"L(?P<laserID>\d{2})A(?P<AID>\d{2})Z(?P<sliceID>\d{2})C(?P<chanID>\d{2})"
    r"\.(?:tif|tiff)$"
)

CQ1 = (
    r"W(?P<wellID>.*)F(?P<fieldID>.*)T(?P<timeID>.*)"
    r"Z(?P<sliceID>.*)C(?P<chanID>.*)\.(?:tif|tiff|png|jpg|jpeg)$"
)

CANONICAL = (
    r"(?P<plateID>[^_]+)_(?P<wellID>[A-Z]\d{2})_"
    r"F(?P<fieldID>\d+)_C(?P<chanID>\d+)\.(?:tif|tiff|png)$"
)
CANONICAL_WITH_TIME = (
    r"(?P<plateID>[^_]+)_(?P<wellID>[A-Z]\d{2})_"
    r"F(?P<fieldID>\d+)_T(?P<timeID>\d+)_C(?P<chanID>\d+)\.(?:tif|tiff|png)$"
)

BUILTIN_REGEXES: Dict[str, str] = {
    "cellvoyager":         CELLVOYAGER,
    "cq1":                 CQ1,
    "yokogawa":            YOKOGAWA,
    "canonical":           CANONICAL,
    "canonical_timelapse": CANONICAL_WITH_TIME,
}



#: For a multi-channel plate spaCR needs at least a channel field
#: PLUS either a well or a field id (usually both).
REQUIRED_MULTICHANNEL: Set[str] = {"chanID"}

#: For a single-channel dataset a field id alone is enough.
REQUIRED_SINGLECHANNEL: Set[str] = {"fieldID"}

#: The "location" fields — wellID and fieldID. At least one required
#: for multi-channel data.
LOCATION_FIELDS: Set[str] = {"wellID", "fieldID"}

#: Every field name spaCR downstream code understands.
KNOWN_FIELDS: Tuple[str, ...] = (
    "plateID", "wellID", "fieldID", "chanID",
    "timeID", "sliceID", "laserID", "AID",
)



@dataclass
[docs] class MetadataRecord: """One parsed filename. :ivar filename: bare filename (no path). :ivar groups: mapping of regex group name → matched text. """ filename: str groups: Dict[str, str]
[docs] def get(self, name: str, default: str = "") -> str: """One captured group, or ``default`` when the pattern did not capture it. :param name: the group's name. :param default: what to return when it is absent. :returns: the captured text. """ return self.groups.get(name, default)
[docs] def apply_regex( filenames: Sequence[str], pattern: str, ) -> Tuple[List[MetadataRecord], List[str]]: """Run ``pattern`` over each filename and return matched records. :param filenames: bare filenames (no directory). :param pattern: regex string; anchored with ``re.match`` semantics. :returns: ``(records, non_matching_filenames)``. """ try: rx = re.compile(pattern) except re.error: return [], list(filenames) records: List[MetadataRecord] = [] missed: List[str] = [] for name in filenames: m = rx.match(name) if m is None: missed.append(name) continue records.append(MetadataRecord(name, dict(m.groupdict()))) return records, missed
[docs] def validate_records( records: Sequence[MetadataRecord], multichannel: bool = True, ) -> List[str]: """Check parsed records against spaCR's downstream requirements. :param records: output of :func:`apply_regex`. :param multichannel: True if the dataset has more than one channel; False for single-channel data (relaxes the requirement set). :returns: list of warning strings — empty when everything is fine. """ if not records: return ["No filenames matched the regex."] warnings: List[str] = [] all_group_names: Set[str] = set() for r in records: for k, v in r.groups.items(): if v: all_group_names.add(k) if multichannel: if "chanID" not in all_group_names: warnings.append( "Missing required field: chanID (multi-channel data " "needs the regex to capture the channel number)." ) if not (all_group_names & LOCATION_FIELDS): warnings.append( "Missing location field: at least one of wellID or " "fieldID must be captured to map each image to a " "well / field." ) else: if "fieldID" not in all_group_names: warnings.append( "Missing required field: fieldID (single-channel data " "still needs a field id to distinguish images)." ) if "plateID" not in all_group_names: warnings.append( "Optional: no plateID captured. spaCR will name the " "generated stack `plate1` by default; edit the regex to " "capture a plate id if you have one." ) return warnings
#: The metadata fields the import actually reads, by name. A proposal that #: captures none of them says nothing about a file whatever it matches. _ROLE_NAMES = frozenset({"plateID", "wellID", "fieldID", "chanID", "timeID", "sliceID"}) def _group_names(pattern: str) -> tuple: """The named groups in ``pattern``, or ``()`` if it does not compile.""" try: return tuple(re.compile(str(pattern or "")).groupindex) except re.error: return ()
[docs] def auto_detect_regex( filenames: Sequence[str], ) -> Tuple[Optional[str], str, int]: """Return the best-fitting regex for a set of filenames. Strategy: 1. Try every :data:`BUILTIN_REGEXES` pattern; if one matches every file it wins immediately. 2. Otherwise pick the built-in that matches the MOST files (>=50 %). 3. If nothing crosses the 50 % bar, synthesise a fresh regex from the common shape of the sample (see :func:`_synthesise_regex`). :param filenames: sample filenames to fit against. :returns: ``(pattern_or_None, label, n_matches)``. ``pattern_or_None`` is None only when synthesis also fails. """ n = len(filenames) if n == 0: return None, "empty", 0 best_label = "none" best_pattern: Optional[str] = None best_hits = -1 for label, pattern in BUILTIN_REGEXES.items(): try: rx = re.compile(pattern) except re.error: continue hits = sum(1 for f in filenames if rx.match(f)) if hits == n: return pattern, label, n if hits > best_hits: best_hits = hits best_label = label best_pattern = pattern if best_hits >= n / 2: return best_pattern, best_label, best_hits try: from ..regex_infer import propose for proposal in propose(filenames): named = {n for n in _group_names(proposal.pattern) if n in _ROLE_NAMES} if named and proposal.matched > best_hits: return proposal.pattern, "inferred", proposal.matched except Exception: # noqa: BLE001 pass synth = _synthesise_regex(filenames) if synth is None: return best_pattern, best_label, best_hits try: synth_rx = re.compile(synth) except re.error: return best_pattern, best_label, best_hits synth_hits = sum(1 for f in filenames if synth_rx.match(f)) if synth_hits < best_hits: return best_pattern, best_label, best_hits return synth, "synthesised", synth_hits
def _synthesise_regex(filenames: Sequence[str]) -> Optional[str]: r"""Best-effort: build a regex from the common shape of filenames. Recognises Illumina/Yokogawa-style tokens:: <letter><digits> → single-letter prefix + digits (F00013, C02, Z01, T0001, etc.) <letter>\d{2} → likely wellID (A01, B12, ...). [A-Z]\d{3} → also wellID plus a big serial. [_-] → literal separators. Strategy: pick ONE filename as a template, walk char-by-char, and replace every digit run with ``\d+``, every letter-prefix + digits combo with a named group when the prefix maps to a known tag. """ if not filenames: return None template = min(filenames, key=lambda s: (len(s), s)) prefix_map = { "F": "fieldID", "T": "timeID", "C": "chanID", "Z": "sliceID", "L": "laserID", "A": "AID", } parts: List[str] = [] used_groups: Set[str] = set() stem, dot, suffix = template.rpartition(".") if not dot: return None tokens = re.split(r"([_-])", stem) for tok in tokens: if tok in ("_", "-"): parts.append(re.escape(tok)) continue wm = re.fullmatch(r"([A-Za-z])(\d{2,3})", tok) if wm and "wellID" not in used_groups: parts.append(r"(?P<wellID>[A-Z]\d{2,3})") used_groups.add("wellID") continue m = re.fullmatch(r"([A-Za-z])(\d+)", tok) if m and m.group(1).upper() in prefix_map: gname = prefix_map[m.group(1).upper()] if gname not in used_groups: parts.append(f"{re.escape(m.group(1))}(?P<{gname}>\\d+)") used_groups.add(gname) continue multi = re.findall(r"([A-Za-z])(\d+)", tok) if multi and all(mp[0].upper() in prefix_map for mp in multi): for letter, digits in multi: gname = prefix_map[letter.upper()] if gname in used_groups: parts.append(f"{re.escape(letter)}\\d+") else: parts.append(f"{re.escape(letter)}(?P<{gname}>\\d+)") used_groups.add(gname) continue if m: parts.append(f"{re.escape(m.group(1))}\\d+") continue if re.fullmatch(r"[A-Za-z0-9]+", tok) and "plateID" not in used_groups: parts.append(r"(?P<plateID>[A-Za-z0-9]+)") used_groups.add("plateID") continue if tok.isdigit(): parts.append(r"\d+") continue parts.append(re.escape(tok)) exts = r"(?:tif|tiff|png|jpg|jpeg)$" return "".join(parts) + r"\." + exts
[docs] def tabulate_records( records: Sequence[MetadataRecord], columns: Optional[Sequence[str]] = None, max_rows: int = 10, random_sample: bool = True, seed: int = 42, ) -> str: """Render a small aligned-column table of records for the Console. :param records: list of parsed records. :param columns: which group names to include; auto-inferred from the first record when None. :param max_rows: cap on rows; sampled at random if exceeded. :param random_sample: True → pick ``max_rows`` at random when the list is longer; False → take the first ``max_rows``. :param seed: RNG seed so the sample is reproducible between runs. :returns: multi-line string ready to feed to :py:meth:`ConsolePanel.append_stdout`. """ if not records: return "(no records — regex did not match any files)" if columns is None: found: Set[str] = set() for r in records: found.update(r.groups.keys()) columns = [f for f in KNOWN_FIELDS if f in found] columns += sorted(found - set(KNOWN_FIELDS)) columns = list(columns) + ["filename"] if len(records) > max_rows: if random_sample: rng = random.Random(seed) sample = rng.sample(list(records), max_rows) else: sample = list(records[:max_rows]) else: sample = list(records) widths = {c: max(len(c), max( (len(_render_cell(r, c)) for r in sample), default=len(c) )) for c in columns} def _row(vals: Sequence[str]) -> str: """One row of the table, padded to the column widths.""" return " " + " ".join(v.ljust(widths[c]) for c, v in zip(columns, vals)) header = _row(columns) rule = " " + " ".join("-" * widths[c] for c in columns) body = [_row([_render_cell(r, c) for c in columns]) for r in sample] return "\n".join([header, rule, *body])
def _render_cell(r: MetadataRecord, col: str) -> str: """Render one cell of the metadata preview. :param r: the parsed record. :param col: the column to show. :returns: the value, or a dash for an axis this filename does not carry. """ if col == "filename": return r.filename return r.get(col, "—")