Skip to content

Object tables and review

See Reviewing and correcting results for the workflow.

patchworks.measure_objects(labels: Any, *, images: Mapping[str, Any] | None = None, axes: str | None = None, pixel_size: Mapping[str, float] | None = None, n_workers: int | None = None) -> dict[str, np.ndarray]

One row per object: size, centroid, bounding box, intensities.

Reads labels once, block by block (its own chunk grid), and never holds more than one block per worker in memory; objects spanning blocks are reassembled exactly.

Parameters:

Name Type Description Default
labels Array or array - like

Integer labels, 0 = background.

required
images mapping of str to array-like

Intensity images of the same shape, by name. Each adds mean_intensity_<name> and std_intensity_<name> columns -- and one more full read of that image, so only ask for what you use.

None
axes str

One letter per axis of labels (default "zyx"-style by ndim).

None
pixel_size mapping

Micrometres per voxel by axis letter. Adds area_um3 (or area_um2) and centroid_<axis>_um columns.

None
n_workers int

Blocks read in parallel (default: the CPUs this job was given).

None

Returns:

Type Description
dict of str to np.ndarray

Columns, all aligned: label, area_voxels, centroid_<axis>, bbox_min_<axis>, bbox_max_<axis> (inclusive, voxel indices), cov_<a><b> (the spread of the object's voxels, in voxels squared: its size and orientation, see :func:shape_columns), plus the optional ones above. Column names follow napari-chunked-regionprops, so either can read the other's tables.

Examples:

>>> lab = np.zeros((1, 4, 6), "int32"); lab[0, 1:3, 1:3] = 1; lab[0, 0, 5] = 7
>>> t = measure_objects(lab)
>>> t["label"].tolist(), t["area_voxels"].tolist()
([1, 7], [4, 1])
>>> t["bbox_max_x"].tolist(), t["centroid_y"].tolist()
([2, 5], [1.5, 0.0])
Source code in src/patchworks/_tables.py
def measure_objects(
    labels: Any,
    *,
    images: Mapping[str, Any] | None = None,
    axes: str | None = None,
    pixel_size: Mapping[str, float] | None = None,
    n_workers: int | None = None,
) -> dict[str, np.ndarray]:
    """One row per object: size, centroid, bounding box, intensities.

    Reads *labels* once, block by block (its own chunk grid), and never holds
    more than one block per worker in memory; objects spanning blocks are
    reassembled exactly.

    Parameters
    ----------
    labels : zarr.Array or array-like
        Integer labels, 0 = background.
    images : mapping of str to array-like, optional
        Intensity images of the same shape, by name. Each adds
        ``mean_intensity_<name>`` and ``std_intensity_<name>`` columns --
        and one more full read of that image, so only ask for what you use.
    axes : str, optional
        One letter per axis of *labels* (default ``"zyx"``-style by ndim).
    pixel_size : mapping, optional
        Micrometres per voxel by axis letter. Adds ``area_um3`` (or
        ``area_um2``) and ``centroid_<axis>_um`` columns.
    n_workers : int, optional
        Blocks read in parallel (default: the CPUs this job was given).

    Returns
    -------
    dict of str to np.ndarray
        Columns, all aligned: ``label``, ``area_voxels``,
        ``centroid_<axis>``, ``bbox_min_<axis>``, ``bbox_max_<axis>``
        (inclusive, voxel indices), ``cov_<a><b>`` (the spread of the
        object's voxels, in voxels squared: its size and orientation, see
        :func:`shape_columns`), plus the optional ones above. Column
        names follow napari-chunked-regionprops, so either can read the
        other's tables.

    Examples
    --------
    >>> lab = np.zeros((1, 4, 6), "int32"); lab[0, 1:3, 1:3] = 1; lab[0, 0, 5] = 7
    >>> t = measure_objects(lab)
    >>> t["label"].tolist(), t["area_voxels"].tolist()
    ([1, 7], [4, 1])
    >>> t["bbox_max_x"].tolist(), t["centroid_y"].tolist()
    ([2, 5], [1.5, 0.0])
    """
    from .plugins.ome_zarr import _default_axes

    images = dict(images or {})
    ndim = len(labels.shape)
    axes = axes or _default_axes(ndim)
    chunk_shape = getattr(labels, "chunks", None)
    if not chunk_shape or not isinstance(chunk_shape[0], (int, np.integer)):
        chunk_shape = tuple(min(s, 256) for s in labels.shape)
    for name, img in images.items():
        if tuple(img.shape) != tuple(labels.shape):
            raise ValueError(
                f"image {name!r} has shape {tuple(img.shape)}, labels "
                f"{tuple(labels.shape)}: measure at the level the labels "
                "were segmented at"
            )
    blocks = list(chunk_slices(labels.shape, chunk_shape))
    nw = n_workers or cpu_allocation()
    img_list = list(images.values())

    def _one(sl):
        lab = np.asarray(labels[sl])
        if not lab.any():
            return None
        return _block_partial(
            lab,
            tuple(s.start for s in sl),
            [np.asarray(img[sl]) for img in img_list],
        )

    started = last = _time.monotonic()
    parts = []
    with ThreadPoolExecutor(max_workers=nw) as ex:
        futures = [ex.submit(_one, sl) for sl in blocks]
        for done, fut in enumerate(as_completed(futures), start=1):
            part = fut.result()
            if part is not None:
                parts.append(part)
            now = _time.monotonic()
            if now - last >= _PROGRESS_INTERVAL_S or done == len(blocks):
                log_progress("measure_objects", done, len(blocks), started)
                last = now

    cols: dict[str, np.ndarray] = {}
    if not parts:
        cols["label"] = np.empty(0, np.int64)
        cols["area_voxels"] = np.empty(0, np.int64)
        for ax in axes:
            cols[f"centroid_{ax}"] = np.empty(0)
        for ax in axes:
            cols[f"bbox_min_{ax}"] = np.empty(0, np.int64)
            cols[f"bbox_max_{ax}"] = np.empty(0, np.int64)
        for i, j in _pairs(ndim):
            cols[f"cov_{axes[i]}{axes[j]}"] = np.empty(0)
        return cols
    m = _merge(parts, len(img_list))
    count = m["count"]
    cols["label"] = m["label"]
    cols["area_voxels"] = count
    for i, ax in enumerate(axes):
        cols[f"centroid_{ax}"] = m["mean"][:, i]
    for i, ax in enumerate(axes):
        cols[f"bbox_min_{ax}"] = m["lo"][:, i]
        cols[f"bbox_max_{ax}"] = m["hi"][:, i]
    for k, (i, j) in enumerate(_pairs(ndim)):
        cols[f"cov_{axes[i]}{axes[j]}"] = m["m2"][:, k] / count
    for i, name in enumerate(images):
        mean = m[f"sum{i}"] / count
        cols[f"mean_intensity_{name}"] = mean
        cols[f"std_intensity_{name}"] = np.sqrt(
            np.maximum(m[f"sq{i}"] / count - mean**2, 0.0)
        )
    if pixel_size:
        size = [float(pixel_size.get(ax, 1.0)) for ax in axes]
        key = "area_um3" if ndim == 3 else f"area_um{ndim}"
        cols[key] = count * float(np.prod(size))
        for ax, s in zip(axes, size):
            cols[f"centroid_{ax}_um"] = cols[f"centroid_{ax}"] * s
    return cols

patchworks.compute_table(store: Union[str, Path], name: str, *, channels: Sequence[int] | None = None, n_workers: int | None = None) -> dict[str, np.ndarray]

Measure labels/<name> of store and store the result as its table.

Parameters:

Name Type Description Default
store str or Path

OME-ZARR image store holding labels/<name>.

required
name str

Label image name.

required
channels sequence of int

Image channels to add mean/std intensity columns for. Read at the pyramid level the labels were segmented at (from their provenance).

None
n_workers int

Parallel block reads.

None

Returns:

Type Description
dict

The columns written.

Source code in src/patchworks/_tables.py
def compute_table(
    store: Union[str, Path],
    name: str,
    *,
    channels: Sequence[int] | None = None,
    n_workers: int | None = None,
) -> dict[str, np.ndarray]:
    """Measure ``labels/<name>`` of *store* and store the result as its table.

    Parameters
    ----------
    store : str or Path
        OME-ZARR image store holding ``labels/<name>``.
    name : str
        Label image name.
    channels : sequence of int, optional
        Image channels to add mean/std intensity columns for. Read at the
        pyramid level the labels were segmented at (from their provenance).
    n_workers : int, optional
        Parallel block reads.

    Returns
    -------
    dict
        The columns written.
    """
    from ._io import load_ome_zarr
    from .plugins.ome_zarr import read_pixel_size

    path = _label_group(store, name)
    grp = zarr.open_group(path, mode="r")
    arr = _level0(grp)
    axes = _spatial_axes(grp, arr.ndim)
    prov = dict(grp.attrs).get(PROVENANCE_KEY) or {}
    settings = prov.get("settings") or {}
    level = int(settings.get("level") or 0)
    images = {}
    for c in channels or ():
        images[f"ch{c}"] = load_ome_zarr(store, channel=int(c), level=level)
    logger.info(
        "measuring labels/%s (%s, %s channel(s))",
        name,
        "x".join(map(str, arr.shape)),
        len(images),
    )
    cols = measure_objects(
        arr,
        images=images,
        axes=axes,
        pixel_size=read_pixel_size(path),
        n_workers=n_workers,
    )
    tile = settings.get("tile_shape")
    write_table(
        path,
        cols,
        attrs={
            "axes": axes,
            "pixel_size": read_pixel_size(path),
            "tile_shape": list(tile) if tile else None,
            "channels": list(images),
        },
    )
    return cols

patchworks.relate_tables(store: Union[str, Path], child: str, parent: str, *, matches: Mapping[int, Mapping[str, float]] | None = None, max_distance_um: float | None = None, n_workers: int | None = None) -> dict[int, dict[str, float]]

Add child's parent columns: which parent object each one is in.

Adds <parent>_id (0 = in none), <parent>_overlap (fraction of the child inside it) and <parent>_overlap_voxels to the child's table, computing the child's table first if it has none.

With max_distance_um, a child touching no parent gets the nearest one within that distance instead (a cilium beside its cell rather than on it), and a <parent>_distance_um column: 0 when overlapping, the gap for those assigned by distance, NaN for none within reach.

Parameters:

Name Type Description Default
store str or Path

OME-ZARR store holding both label images.

required
child str

Label image names, e.g. "cilia_labels" in "cyto_labels".

required
parent str

Label image names, e.g. "cilia_labels" in "cyto_labels".

required
matches mapping

:func:~patchworks.label_relations output for this pair, when already computed -- otherwise it is computed here (a read of both).

None
n_workers int

Parallel chunk reads.

None

Returns:

Type Description
dict

The matches used.

Source code in src/patchworks/_tables.py
def relate_tables(
    store: Union[str, Path],
    child: str,
    parent: str,
    *,
    matches: Mapping[int, Mapping[str, float]] | None = None,
    max_distance_um: float | None = None,
    n_workers: int | None = None,
) -> dict[int, dict[str, float]]:
    """Add *child*'s parent columns: which *parent* object each one is in.

    Adds ``<parent>_id`` (0 = in none), ``<parent>_overlap`` (fraction of
    the child inside it) and ``<parent>_overlap_voxels`` to the child's
    table, computing the child's table first if it has none.

    With *max_distance_um*, a child touching no parent gets the **nearest**
    one within that distance instead (a cilium beside its cell rather than
    on it), and a ``<parent>_distance_um`` column: 0 when overlapping, the
    gap for those assigned by distance, NaN for none within reach.

    Parameters
    ----------
    store : str or Path
        OME-ZARR store holding both label images.
    child, parent : str
        Label image names, e.g. ``"cilia_labels"`` in ``"cyto_labels"``.
    matches : mapping, optional
        :func:`~patchworks.label_relations` output for this pair, when
        already computed -- otherwise it is computed here (a read of both).
    n_workers : int, optional
        Parallel chunk reads.

    Returns
    -------
    dict
        The matches used.
    """
    import dask.array as da

    from ._relations import label_relations

    cpath, ppath = _label_group(store, child), _label_group(store, parent)
    if not has_table(cpath) or is_stale(cpath):
        compute_table(store, child, n_workers=n_workers)
    if matches is None:
        a = da.from_zarr(_level0(zarr.open_group(cpath, mode="r")))
        b = da.from_zarr(_level0(zarr.open_group(ppath, mode="r")))
        a, b = align_chunks(a, b)
        matches = label_relations(a, b, n_workers=n_workers)
    labels = zarr.open_group(f"{cpath}/{TABLE_GROUP}", mode="r")["label"][...]
    pid = np.zeros(labels.size, np.int64)
    frac = np.zeros(labels.size)
    vox = np.zeros(labels.size, np.int64)
    for i, lab in enumerate(labels.tolist()):
        m = matches.get(lab)
        if m is not None:
            pid[i], frac[i], vox[i] = (
                m["match"],
                m["overlap_fraction"],
                m["overlap_voxels"],
            )
    cols = {
        f"{parent}_id": pid,
        f"{parent}_overlap": frac,
        f"{parent}_overlap_voxels": vox,
    }
    if max_distance_um is not None:
        dist = np.where(pid != 0, 0.0, np.nan)
        orphans = np.flatnonzero(pid == 0)
        found = nearest_parents(
            store,
            child,
            parent,
            labels[orphans],
            max_distance_um=float(max_distance_um),
            n_workers=n_workers,
        )
        for i, lab in zip(orphans, labels[orphans].tolist()):
            if lab in found:
                pid[i], dist[i] = found[lab]
        cols[f"{parent}_distance_um"] = dist
    add_columns(cpath, cols)
    return dict(matches)

patchworks._tables.nearest_parents(store: Union[str, Path], child: str, parent: str, labels: Sequence[int], *, max_distance_um: float, n_workers: int | None = None) -> dict[int, tuple[int, float]]

The nearest parent object to each of labels (objects of child), within max_distance_um: {label: (parent_id, distance_um)}.

Exact surface-to-surface distance in µm (anisotropic voxels included), from the voxels around each object only: its bounding box grown by the distance. Meant for the few objects touching no parent, not all.

Source code in src/patchworks/_tables.py
def nearest_parents(
    store: Union[str, Path],
    child: str,
    parent: str,
    labels: Sequence[int],
    *,
    max_distance_um: float,
    n_workers: int | None = None,
) -> dict[int, tuple[int, float]]:
    """The nearest *parent* object to each of *labels* (objects of *child*),
    within *max_distance_um*: ``{label: (parent_id, distance_um)}``.

    Exact surface-to-surface distance in µm (anisotropic voxels included),
    from the voxels around each object only: its bounding box grown by
    the distance. Meant for the few objects touching no parent, not all.
    """
    from ._io import open_group_any
    from .plugins.ome_zarr import read_pixel_size

    cpath, ppath = _label_group(store, child), _label_group(store, parent)
    cgroup = open_group_any(cpath)
    carr, parr = _level0(cgroup), _level0(open_group_any(ppath))
    axes = _spatial_axes(cgroup, carr.ndim)
    size = read_pixel_size(cpath)
    spacing = np.array([float(size.get(ax, 1.0)) for ax in axes])
    margin = np.ceil(max_distance_um / spacing).astype(int)
    table = zarr.open_group(f"{cpath}/{TABLE_GROUP}", mode="r")
    all_labels = table["label"][...]
    lo = np.stack([np.asarray(table[f"bbox_min_{ax}"][...]) for ax in axes], 1)
    hi = np.stack([np.asarray(table[f"bbox_max_{ax}"][...]) for ax in axes], 1)
    where = {int(v): i for i, v in enumerate(all_labels)}

    def one(label: int):
        i = where.get(int(label))
        if i is None:
            return None
        a = np.maximum(lo[i] - margin, 0)
        b = np.minimum(hi[i] + margin + 1, carr.shape)
        sl = tuple(slice(int(x), int(y)) for x, y in zip(a, b))
        mask = np.asarray(carr[sl]) == label
        region = np.asarray(parr[sl])
        if not region.any() or not mask.any():
            return None
        dist, idx = ndi.distance_transform_edt(
            region == 0, sampling=spacing, return_indices=True
        )
        pos = np.argmin(np.where(mask, dist, np.inf))
        d = float(dist.flat[pos])
        if d > max_distance_um:
            return None
        nearest = tuple(ix.flat[pos] for ix in idx)
        return int(label), (int(region[nearest]), d)

    with ThreadPoolExecutor(max_workers=n_workers or cpu_allocation()) as ex:
        return dict(r for r in ex.map(one, labels) if r is not None)

patchworks._tables.shape_columns(df: 'pd.DataFrame', axes: str, pixel_size: Mapping[str, float] | None) -> dict[str, np.ndarray]

Length, elongation and main axis of each object, from its moments.

length is the extent along the main axis of a uniform rod with the same spread (sqrt(12 * largest variance)): exact for a straight rod, shorter than the path of a curved one. elongation is the ratio of the two largest spreads (1: round, large: rod-like). axis_<a> is the main axis as a unit vector (sign fixed so its first non-zero component is positive). In µm when pixel_size is known.

Examples:

>>> import pandas as pd
>>> rod = pd.DataFrame({"cov_zz": [0.0], "cov_yy": [0.0], "cov_xx": [12.0],
...     "cov_zy": [0.0], "cov_zx": [0.0], "cov_yx": [0.0]})
>>> cols = shape_columns(rod, "zyx", {"z": 1, "y": 1, "x": 0.5})
>>> round(float(cols["length_um"][0]), 3), float(cols["axis_x"][0])
(6.0, 1.0)
Source code in src/patchworks/_tables.py
def shape_columns(
    df: "pd.DataFrame", axes: str, pixel_size: Mapping[str, float] | None
) -> dict[str, np.ndarray]:
    """Length, elongation and main axis of each object, from its moments.

    ``length`` is the extent along the main axis of a uniform rod with the
    same spread (``sqrt(12 * largest variance)``): exact for a straight
    rod, shorter than the path of a curved one. ``elongation`` is the ratio
    of the two largest spreads (1: round, large: rod-like). ``axis_<a>``
    is the main axis as a unit vector (sign fixed so its first non-zero
    component is positive). In µm when *pixel_size* is known.

    Examples
    --------
    >>> import pandas as pd
    >>> rod = pd.DataFrame({"cov_zz": [0.0], "cov_yy": [0.0], "cov_xx": [12.0],
    ...     "cov_zy": [0.0], "cov_zx": [0.0], "cov_yx": [0.0]})
    >>> cols = shape_columns(rod, "zyx", {"z": 1, "y": 1, "x": 0.5})
    >>> round(float(cols["length_um"][0]), 3), float(cols["axis_x"][0])
    (6.0, 1.0)
    """
    cov = physical_cov(df, axes, pixel_size)
    unit = "um" if pixel_size else "voxels"
    if not len(df) or np.isnan(cov).all():
        return {}
    w, v = np.linalg.eigh(cov)  # ascending
    w = np.clip(w, 0, None)
    main = v[:, :, -1]
    first = np.argmax(np.abs(main) > 1e-9, axis=1)
    sign = np.sign(main[np.arange(len(main)), first])
    main = main * np.where(sign == 0, 1, sign)[:, None]
    out = {f"length_{unit}": np.sqrt(12 * w[:, -1])}
    with np.errstate(divide="ignore", invalid="ignore"):
        out["elongation"] = (
            np.sqrt(w[:, -1] / w[:, -2]) if w.shape[1] > 1 else np.ones(len(w))
        )
    for k, ax in enumerate(axes):
        out[f"axis_{ax}"] = main[:, k]
    return out

patchworks.read_table(group: Union[str, Path], *, check: bool = True) -> 'pd.DataFrame'

The table of label group group as a DataFrame indexed by label.

Examples:

>>> read_table("scan.zarr/labels/cilia_labels")
Source code in src/patchworks/_tables.py
def read_table(
    group: Union[str, Path], *, check: bool = True
) -> "pd.DataFrame":
    """The table of label group *group* as a DataFrame indexed by ``label``.

    Examples
    --------
    >>> read_table("scan.zarr/labels/cilia_labels")  # doctest: +SKIP
    """
    import pandas as pd

    cols = read_columns(group, check=check)
    return pd.DataFrame(cols).set_index("label")

patchworks.Review

The objects of one store, their flags and the reviewer's decisions.

Parameters:

Name Type Description Default
store str or Path

OME-ZARR image store whose label images carry tables.

required
expect mapping

Expected child counts per parent object: {"cyto_labels": {"nuclei_labels": (1, 1), "cilia_labels": (0, 2)}} -- a cell with no nucleus or 3 cilia is then flagged. Falls back to what the tables' metadata recorded (the workflow's review:).

None
min_overlap float or mapping

A child less than this fraction inside its parent is flagged (default 0.5; per {child: {parent: value}} if a mapping).

None
seam_tolerance int

Voxels of slack when matching two objects meeting at a tile seam.

1
names iterable of str

Only these label images (default: every one with a table).

None
position mapping

Classify objects by where they sit in their parent: {"cilia_labels": {"parent": "cyto_labels", "apical": "nuclei_labels"}} -- see :meth:positions. Falls back to the store's recorded rules.

None

Examples:

>>> rv = Review("scan.zarr", expect={"cells": {"nuclei": (1, 1)}})
>>> rv.flags("nuclei")[:3]
>>> rv.decide("nuclei", 17, "wrong")
>>> rv.summary("nuclei")
Source code in src/patchworks/_review.py
 149
 150
 151
 152
 153
 154
 155
 156
 157
 158
 159
 160
 161
 162
 163
 164
 165
 166
 167
 168
 169
 170
 171
 172
 173
 174
 175
 176
 177
 178
 179
 180
 181
 182
 183
 184
 185
 186
 187
 188
 189
 190
 191
 192
 193
 194
 195
 196
 197
 198
 199
 200
 201
 202
 203
 204
 205
 206
 207
 208
 209
 210
 211
 212
 213
 214
 215
 216
 217
 218
 219
 220
 221
 222
 223
 224
 225
 226
 227
 228
 229
 230
 231
 232
 233
 234
 235
 236
 237
 238
 239
 240
 241
 242
 243
 244
 245
 246
 247
 248
 249
 250
 251
 252
 253
 254
 255
 256
 257
 258
 259
 260
 261
 262
 263
 264
 265
 266
 267
 268
 269
 270
 271
 272
 273
 274
 275
 276
 277
 278
 279
 280
 281
 282
 283
 284
 285
 286
 287
 288
 289
 290
 291
 292
 293
 294
 295
 296
 297
 298
 299
 300
 301
 302
 303
 304
 305
 306
 307
 308
 309
 310
 311
 312
 313
 314
 315
 316
 317
 318
 319
 320
 321
 322
 323
 324
 325
 326
 327
 328
 329
 330
 331
 332
 333
 334
 335
 336
 337
 338
 339
 340
 341
 342
 343
 344
 345
 346
 347
 348
 349
 350
 351
 352
 353
 354
 355
 356
 357
 358
 359
 360
 361
 362
 363
 364
 365
 366
 367
 368
 369
 370
 371
 372
 373
 374
 375
 376
 377
 378
 379
 380
 381
 382
 383
 384
 385
 386
 387
 388
 389
 390
 391
 392
 393
 394
 395
 396
 397
 398
 399
 400
 401
 402
 403
 404
 405
 406
 407
 408
 409
 410
 411
 412
 413
 414
 415
 416
 417
 418
 419
 420
 421
 422
 423
 424
 425
 426
 427
 428
 429
 430
 431
 432
 433
 434
 435
 436
 437
 438
 439
 440
 441
 442
 443
 444
 445
 446
 447
 448
 449
 450
 451
 452
 453
 454
 455
 456
 457
 458
 459
 460
 461
 462
 463
 464
 465
 466
 467
 468
 469
 470
 471
 472
 473
 474
 475
 476
 477
 478
 479
 480
 481
 482
 483
 484
 485
 486
 487
 488
 489
 490
 491
 492
 493
 494
 495
 496
 497
 498
 499
 500
 501
 502
 503
 504
 505
 506
 507
 508
 509
 510
 511
 512
 513
 514
 515
 516
 517
 518
 519
 520
 521
 522
 523
 524
 525
 526
 527
 528
 529
 530
 531
 532
 533
 534
 535
 536
 537
 538
 539
 540
 541
 542
 543
 544
 545
 546
 547
 548
 549
 550
 551
 552
 553
 554
 555
 556
 557
 558
 559
 560
 561
 562
 563
 564
 565
 566
 567
 568
 569
 570
 571
 572
 573
 574
 575
 576
 577
 578
 579
 580
 581
 582
 583
 584
 585
 586
 587
 588
 589
 590
 591
 592
 593
 594
 595
 596
 597
 598
 599
 600
 601
 602
 603
 604
 605
 606
 607
 608
 609
 610
 611
 612
 613
 614
 615
 616
 617
 618
 619
 620
 621
 622
 623
 624
 625
 626
 627
 628
 629
 630
 631
 632
 633
 634
 635
 636
 637
 638
 639
 640
 641
 642
 643
 644
 645
 646
 647
 648
 649
 650
 651
 652
 653
 654
 655
 656
 657
 658
 659
 660
 661
 662
 663
 664
 665
 666
 667
 668
 669
 670
 671
 672
 673
 674
 675
 676
 677
 678
 679
 680
 681
 682
 683
 684
 685
 686
 687
 688
 689
 690
 691
 692
 693
 694
 695
 696
 697
 698
 699
 700
 701
 702
 703
 704
 705
 706
 707
 708
 709
 710
 711
 712
 713
 714
 715
 716
 717
 718
 719
 720
 721
 722
 723
 724
 725
 726
 727
 728
 729
 730
 731
 732
 733
 734
 735
 736
 737
 738
 739
 740
 741
 742
 743
 744
 745
 746
 747
 748
 749
 750
 751
 752
 753
 754
 755
 756
 757
 758
 759
 760
 761
 762
 763
 764
 765
 766
 767
 768
 769
 770
 771
 772
 773
 774
 775
 776
 777
 778
 779
 780
 781
 782
 783
 784
 785
 786
 787
 788
 789
 790
 791
 792
 793
 794
 795
 796
 797
 798
 799
 800
 801
 802
 803
 804
 805
 806
 807
 808
 809
 810
 811
 812
 813
 814
 815
 816
 817
 818
 819
 820
 821
 822
 823
 824
 825
 826
 827
 828
 829
 830
 831
 832
 833
 834
 835
 836
 837
 838
 839
 840
 841
 842
 843
 844
 845
 846
 847
 848
 849
 850
 851
 852
 853
 854
 855
 856
 857
 858
 859
 860
 861
 862
 863
 864
 865
 866
 867
 868
 869
 870
 871
 872
 873
 874
 875
 876
 877
 878
 879
 880
 881
 882
 883
 884
 885
 886
 887
 888
 889
 890
 891
 892
 893
 894
 895
 896
 897
 898
 899
 900
 901
 902
 903
 904
 905
 906
 907
 908
 909
 910
 911
 912
 913
 914
 915
 916
 917
 918
 919
 920
 921
 922
 923
 924
 925
 926
 927
 928
 929
 930
 931
 932
 933
 934
 935
 936
 937
 938
 939
 940
 941
 942
 943
 944
 945
 946
 947
 948
 949
 950
 951
 952
 953
 954
 955
 956
 957
 958
 959
 960
 961
 962
 963
 964
 965
 966
 967
 968
 969
 970
 971
 972
 973
 974
 975
 976
 977
 978
 979
 980
 981
 982
 983
 984
 985
 986
 987
 988
 989
 990
 991
 992
 993
 994
 995
 996
 997
 998
 999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
class Review:
    """The objects of one store, their flags and the reviewer's decisions.

    Parameters
    ----------
    store : str or Path
        OME-ZARR image store whose label images carry tables.
    expect : mapping, optional
        Expected child counts per parent object:
        ``{"cyto_labels": {"nuclei_labels": (1, 1), "cilia_labels": (0, 2)}}``
        -- a cell with no nucleus or 3 cilia is then flagged. Falls back to
        what the tables' metadata recorded (the workflow's ``review:``).
    min_overlap : float or mapping, optional
        A child less than this fraction inside its parent is flagged
        (default 0.5; per ``{child: {parent: value}}`` if a mapping).
    seam_tolerance : int, optional
        Voxels of slack when matching two objects meeting at a tile seam.
    names : iterable of str, optional
        Only these label images (default: every one with a table).
    position : mapping, optional
        Classify objects by where they sit in their parent:
        ``{"cilia_labels": {"parent": "cyto_labels", "apical":
        "nuclei_labels"}}`` -- see :meth:`positions`. Falls back to the
        store's recorded rules.

    Examples
    --------
    >>> rv = Review("scan.zarr", expect={"cells": {"nuclei": (1, 1)}})  # doctest: +SKIP
    >>> rv.flags("nuclei")[:3]  # doctest: +SKIP
    >>> rv.decide("nuclei", 17, "wrong")  # doctest: +SKIP
    >>> rv.summary("nuclei")  # doctest: +SKIP
    """

    def __init__(
        self,
        store: Union[str, Path],
        *,
        expect: Mapping[str, Mapping[str, Iterable[int]]] | None = None,
        min_overlap: float | Mapping[str, Mapping[str, float]] | None = None,
        seam_tolerance: int = 1,
        names: Iterable[str] | None = None,
        position: Mapping[str, Mapping[str, Any]] | None = None,
    ) -> None:
        self.store = str(store).rstrip("/")
        self.seam_tolerance = int(seam_tolerance)
        self.names: list[str] = []
        self.meta: dict[str, dict[str, Any]] = {}
        self.tables: dict[str, pd.DataFrame] = {}
        self.stale: list[str] = []
        wanted = None if names is None else set(names)
        for name in table_names(self.store):
            if wanted is not None and name not in wanted:
                continue
            group = _label_group(self.store, name)
            if not has_table(group):
                continue
            if is_stale(group):
                self.stale.append(name)
                continue
            self.names.append(name)
            self.meta[name] = table_meta(group)
            self.tables[name] = read_table(group, check=False)
        if self.stale:
            logger.warning(
                "tables of %s were computed from older labels and are "
                "ignored; recompute them (patchworks tables %s)",
                ", ".join(self.stale),
                self.store,
            )
        # A table's parents are the label images it has relation columns for.
        self.parents = {
            n: [
                p
                for p in self.names
                if p != n
                and f"{p}_id" in self.tables[n]
                and f"{p}_overlap" in self.tables[n]
            ]
            for n in self.names
        }
        self.children = {
            n: [c for c in self.names if n in self.parents[c]]
            for n in self.names
        }
        rules = read_rules(self.store)
        self.expect: dict[str, dict[str, tuple[int, int]]] = {}
        for n, rules_n in (rules.get("expect") or {}).items():
            for child, rng in rules_n.items():
                self.expect.setdefault(n, {})[child] = _range(rng)
        self._recorded_overlap = rules.get("min_overlap")
        for n, rules in (expect or {}).items():
            for child, rng in rules.items():
                self.expect.setdefault(n, {})[child] = _range(rng)
        self._min_overlap = min_overlap
        self.position: dict[str, dict[str, Any]] = {
            k: dict(v) for k, v in (rules.get("position") or {}).items()
        }
        for child, rule in (position or {}).items():
            self.position[child] = dict(rule)
        for child, rule in list(self.position.items()):
            if child not in self.tables or "parent" not in rule:
                logger.warning("position rule for %s ignored: %s", child, rule)
                del self.position[child]
        self.decisions: dict[str, dict[int, dict[str, Any]]] = {}
        self.seed: dict[str, int] = {}
        for n in self.names:
            self._load_decisions(n)
        self._cache: dict[str, pd.DataFrame] = {}
        self._flag_cache: dict[str, tuple[pd.DataFrame, pd.DataFrame]] = {}

    # -- decisions ---------------------------------------------------------

    def _table_group(self, name: str, mode: str = "r"):
        from ._io import open_group_any

        return open_group_any(
            f"{_label_group(self.store, name)}/{TABLE_GROUP}", mode=mode
        )

    def _load_decisions(self, name: str) -> None:
        log = dict(self._table_group(name).attrs.get(REVIEW_KEY) or {})
        fingerprint = label_fingerprint(
            zarr.open_group(_label_group(self.store, name), mode="r")
        )
        decisions = log.get("decisions") or {}
        if decisions and log.get("labels") != fingerprint:
            logger.warning(
                "%s: the review decisions were made on older labels; "
                "starting a fresh review",
                name,
            )
            decisions = {}
        self.decisions[name] = {int(k): v for k, v in decisions.items()}
        self.seed[name] = int(log.get("seed", 0))

    def _save(self, name: str) -> None:
        try:
            group = self._table_group(name, mode="r+")
        except Exception as exc:
            raise PermissionError(
                f"cannot save review decisions into {self.store} (read-only, "
                "or a .zip bundle?): review a writable copy of the store"
            ) from exc
        group.attrs[REVIEW_KEY] = {
            "updated": _dt.datetime.now(_dt.timezone.utc).isoformat(),
            "labels": label_fingerprint(
                zarr.open_group(_label_group(self.store, name), mode="r")
            ),
            "seed": self.seed[name],
            "decisions": {
                str(k): v for k, v in sorted(self.decisions[name].items())
            },
        }
        self._cache.clear()
        self._flag_cache.clear()

    def decide(
        self,
        name: str,
        label: int,
        action: str,
        *,
        parent: str | None = None,
        parent_id: int | None = None,
        into: int | None = None,
        position: str | None = None,
        queue: str = "flagged",
        reviewer: str | None = None,
    ) -> None:
        """Record a decision about object *label* of label image *name*.

        Parameters
        ----------
        action : {"ok", "wrong", "parent", "merge"}
            ``ok``: correct as it is. ``wrong``: not a real object (dropped
            from the corrected tables). ``parent``: belongs to *parent_id*
            of label image *parent* (0 = to none). ``merge``: the same
            object as *into*. ``position``: its position class is
            *position* (one of :data:`POSITIONS`), whatever was computed.
        queue : str
            The queue it was reviewed from; ``"random"`` decisions feed the
            error estimate.
        """
        if name not in self.tables:
            raise KeyError(f"no table for labels/{name}")
        label = int(label)
        if label not in self.tables[name].index:
            raise KeyError(f"{name} has no object {label}")
        if action not in ACTIONS:
            raise ValueError(f"action must be one of {ACTIONS}, got {action!r}")
        entry: dict[str, Any] = {
            "action": action,
            "queue": queue,
            "time": _dt.datetime.now(_dt.timezone.utc).isoformat(
                timespec="seconds"
            ),
        }
        if reviewer:
            entry["by"] = reviewer
        if action == "parent":
            if parent not in self.parents[name]:
                raise ValueError(
                    f"{name} has no parent image {parent!r} "
                    f"(has {self.parents[name]})"
                )
            pid = int(parent_id or 0)
            if pid and pid not in self.tables[parent].index:
                raise KeyError(f"{parent} has no object {pid}")
            entry.update(parent=parent, parent_id=pid)
            previous = self.decisions[name].get(label, {})
            if previous.get("action") == "parent":
                entry["parents"] = {**previous.get("parents", {}), parent: pid}
            else:
                entry["parents"] = {parent: pid}
        elif action == "position":
            if name not in self.position:
                raise ValueError(f"{name} has no position rule")
            if position not in POSITIONS:
                raise ValueError(f"position must be one of {POSITIONS}")
            entry["position"] = position
        elif action == "merge":
            into = int(into or 0)
            if into == label or into not in self.tables[name].index:
                raise KeyError(f"cannot merge {name} {label} into {into}")
            entry["into"] = into
        self.decisions[name][label] = entry
        self._save(name)

    def relations(self) -> list[tuple[str, str]]:
        """Every (child, parent) pair the tables relate."""
        return [(c, p) for c in self.names for p in self.parents[c]]

    def undo(self, name: str, label: int) -> None:
        """Forget the decision about *label* (it goes back to unreviewed)."""
        if self.decisions[name].pop(int(label), None) is not None:
            self._save(name)

    # -- the corrected view --------------------------------------------------

    def roots(self, name: str) -> dict[int, int]:
        """Where each decided object ends up: 0 if rejected, the object it
        was merged into (followed to the end) otherwise. Undecided objects
        are absent (they map to themselves)."""
        dec = self.decisions[name]
        out: dict[int, int] = {}
        for label in dec:
            seen, cur = {label}, label
            while True:
                d = dec.get(cur)
                if d is None or d["action"] in ("ok", "parent", "position"):
                    break
                if d["action"] == "wrong":
                    cur = 0
                    break
                cur = int(d["into"])
                if cur in seen:  # a merge cycle: keep the smallest id
                    cur = min(seen)
                    break
                seen.add(cur)
            if cur != label:
                out[label] = cur
        return out

    def _corrected(self, name: str) -> "pd.DataFrame":
        """*name*'s rows with the decisions applied, before derived columns
        (counts, shape, position) -- which need other tables' rows, never
        their derived columns, so nothing here recurses."""
        key = ("rows", name)
        if key in self._cache:
            return self._cache[key]
        import pandas as pd

        df = self.tables[name].copy()
        dec = self.decisions[name]
        for p in self.parents[name]:
            col = f"{p}_id"
            for label, d in dec.items():
                if d["action"] == "parent" and p in d.get("parents", {}):
                    df.loc[label, col] = d["parents"][p]
                    df.loc[label, f"{p}_overlap"] = np.nan
            proots = self.roots(p)
            if proots:
                df[col] = _remap(df[col].to_numpy(dtype=np.int64), proots)
        roots = self.roots(name)
        if roots:
            target = pd.Series(df.index, index=df.index)
            target.update(pd.Series(roots))
            df = df[target != 0]
            target = target[target != 0]
            if (target != target.index).any():
                df = _combine(df, target)
        qc = pd.Series("", index=df.index)
        for label, d in dec.items():
            if label in qc.index:
                qc[label] = "ok" if d["action"] == "ok" else "fixed"
        for label, root in roots.items():
            if root and root in qc.index:
                qc[root] = "fixed"
        df["qc"] = qc
        self._cache[key] = df
        return df

    def effective(self, name: str) -> "pd.DataFrame":
        """*name*'s table with every decision applied.

        Rejected objects are dropped; merged objects become one row (sizes
        added, centroids, spreads and intensities combined exactly, boxes
        joined); parent columns follow reassignments and the parents' own
        merges; ``qc`` says ``ok``, ``fixed`` or ``""`` (not reviewed).
        Derived columns: ``n_<child>`` (children counted), ``length_um``,
        ``elongation``, ``axis_<a>`` (shape, see
        :func:`~patchworks._tables.shape_columns`) and, for label images
        with a position rule, ``position`` and its measures (see
        :meth:`positions`).
        """
        if name in self._cache:
            return self._cache[name]
        from ._tables import shape_columns

        df = self._corrected(name).copy()
        for child in self.children[name]:
            ce = self._corrected(child)
            counts = ce[f"{name}_id"].value_counts()
            df[f"n_{child}"] = (
                counts.reindex(df.index).fillna(0).astype(np.int64)
            )
        meta = self.meta[name]
        for col, values in shape_columns(
            df, meta.get("axes") or "", meta.get("pixel_size")
        ).items():
            df[col] = values
        if name in self.position:
            for col, values in self.positions(name, df).items():
                df[col] = values
        for child in self.children[name]:
            if (
                child in self.position
                and self.position[child]["parent"] == name
            ):
                pos = self.effective(child)
                table = (
                    pos[pos[f"{name}_id"] != 0]
                    .groupby([f"{name}_id", "position"])
                    .size()
                    .unstack(fill_value=0)
                )
                for cls in POSITIONS:
                    counts = table[cls] if cls in table else None
                    df[f"n_{child}_{cls}"] = (
                        counts.reindex(df.index).fillna(0).astype(np.int64)
                        if counts is not None
                        else 0
                    )
        self._cache[name] = df
        return df

    def positions(
        self, name: str, df: "pd.DataFrame | None" = None
    ) -> dict[str, Any]:
        """Where each object of *name* sits in its parent.

        Per parent object, an apical axis: a fixed direction (``"+z"``:
        apical is up the z axis), or pointing away from the parent's
        children of another label image (``"nuclei_labels"``: away from
        the nucleus, for epithelia whose nuclei sit basally), or towards
        them (``"towards:<name>"``). Each object's **base** is its end
        nearer the parent's centre (a cilium grows out from its base).

        The class is the parent surface the base is nearest to -- top
        (``apical``), bottom (``basal``) or side wall (``lateral``) --
        each depth measured relative to the parent's size in that
        direction; ``central`` when deeper than ``central_depth`` (default
        0.5, i.e. half-way) from all three. The parent's shape is taken
        from its moments (an equivalent cylinder), so this is a
        classification, not a surface distance.

        Returns columns ``position`` (the class; ``outside`` with no
        parent; ``unknown`` when the axis cannot be told, e.g. no nucleus),
        ``position_axial`` (-1 basal .. +1 apical, in half-heights of the
        parent along its axis), ``position_radial`` (0 on the axis .. 1 at
        the side) and ``angle_to_axis_deg`` (0: along the apical axis, 90:
        across it). A reviewer's ``position`` decision overrides the class.
        """
        from ._tables import physical_centroids, physical_cov

        rule = self.position[name]
        parent = rule["parent"]
        df = self._corrected(name) if df is None else df
        n = len(df)
        out: dict[str, Any] = {
            "position": np.full(n, "unknown", dtype=object),
            "position_axial": np.full(n, np.nan),
            "position_radial": np.full(n, np.nan),
            "angle_to_axis_deg": np.full(n, np.nan),
        }
        if parent not in self.parents[name] or not n:
            return out
        axes = self.meta[name].get("axes") or ""
        size = self.meta[name].get("pixel_size")
        pe = self._corrected(parent)
        pid = df[f"{parent}_id"].to_numpy(dtype=np.int64)
        spread = [f"cov_{a}{a}" for a in axes]
        missing = [
            t
            for t, d in ((name, df), (parent, pe))
            if not set(spread) <= set(d)
        ]
        if missing:
            logger.warning(
                "no spread (cov_*) columns in the table(s) of %s -- written "
                "by an older patchworks; recompute them (patchworks tables "
                "%s) to classify positions",
                ", ".join(missing),
                self.store,
            )
            out["position"][pid == 0] = "outside"
            return out
        out["position"][pid == 0] = "outside"
        inside = np.flatnonzero(np.isin(pid, pe.index.to_numpy()))
        if not inside.size:
            return out
        prow = pe.index.get_indexer(pid[inside])
        centre = physical_centroids(pe, axes, size)[prow]
        pcov = physical_cov(pe, axes, size)[prow]
        axis = self._apical_axes(name, rule, pe, axes, size)[prow]

        c = physical_centroids(df, axes, size)[inside]
        w, v = np.linalg.eigh(physical_cov(df, axes, size)[inside])
        main = v[:, :, -1]
        half = np.sqrt(12 * np.clip(w[:, -1], 0, None)) / 2
        ends = np.stack([c + half[:, None] * main, c - half[:, None] * main])
        near = np.argmin(np.linalg.norm(ends - centre, axis=2), axis=0)
        base = ends[near, np.arange(inside.size)]

        rel = base - centre
        along = np.einsum("ij,ij->i", rel, axis)
        var_a = np.einsum("ij,ijk,ik->i", axis, pcov, axis)
        axial = along / np.sqrt(3 * np.clip(var_a, 1e-12, None))
        perp = rel - along[:, None] * axis
        across = np.sqrt(
            2 * np.clip(np.trace(pcov, axis1=1, axis2=2) - var_a, 1e-12, None)
        )
        radial = np.linalg.norm(perp, axis=1) / across
        angle = np.degrees(
            np.arccos(np.clip(np.abs(np.einsum("ij,ij->i", main, axis)), 0, 1))
        )
        # The surface the base is nearest to, each depth relative to the
        # parent's size in that direction: top, bottom or side wall. Deeper
        # than `central_depth` from all three: central.
        depth = np.stack([1 - axial, 1 + axial, 1 - radial], 1)
        cls = np.array(["apical", "basal", "lateral"], dtype=object)[
            np.argmin(depth, axis=1)
        ]
        cls[depth.min(axis=1) > float(rule.get("central_depth", 0.5))] = (
            "central"
        )
        known = np.isfinite(axis).all(axis=1)
        cls[~known] = "unknown"
        out["position"][inside] = cls
        out["position_axial"][inside] = np.where(known, axial, np.nan)
        out["position_radial"][inside] = np.where(known, radial, np.nan)
        out["angle_to_axis_deg"][inside] = np.where(known, angle, np.nan)
        for label, d in self.decisions[name].items():
            if d["action"] == "position" and label in df.index:
                out["position"][df.index.get_loc(label)] = d["position"]
        return out

    def _apical_axes(self, name, rule, pe, axes, size) -> np.ndarray:
        """Unit apical direction per parent object (NaN: undetermined)."""
        from ._tables import physical_centroids, physical_cov

        spec = str(rule.get("apical", ""))
        n = len(pe)
        if spec in _AXES_SIGNS:
            ax, sign = _AXES_SIGNS[spec]
            if ax not in axes:
                raise ValueError(f"apical {spec!r}: {name} has axes {axes!r}")
            vec = np.zeros(len(axes))
            vec[axes.index(ax)] = sign
            return np.tile(vec, (n, 1))
        towards = spec.startswith("towards:")
        ref = spec.split(":", 1)[1] if towards else spec
        parent = rule["parent"]
        if ref not in self.tables or parent not in self.parents.get(ref, []):
            raise ValueError(
                f"apical {spec!r} for {name}: needs {ref!r} related to "
                f"{parent!r} (its objects inside the parent objects), or a "
                f"direction like '+z'"
            )
        re = self._corrected(ref)
        re = re[re[f"{parent}_id"].isin(pe.index)]
        weights = re["area_voxels"].to_numpy(dtype=float)
        rc = physical_centroids(re, axes, size) * weights[:, None]
        import pandas as pd

        sums = pd.DataFrame(rc).groupby(re[f"{parent}_id"].to_numpy()).sum()
        wsum = pd.Series(weights).groupby(re[f"{parent}_id"].to_numpy()).sum()
        refc = (sums.div(wsum, axis=0)).reindex(pe.index).to_numpy()
        centre = physical_centroids(pe, axes, size)
        vec = centre - refc if not towards else refc - centre
        norm = np.linalg.norm(vec, axis=1)
        scale = np.sqrt(
            np.trace(physical_cov(pe, axes, size), axis1=1, axis2=2)
        )
        with np.errstate(invalid="ignore", divide="ignore"):
            vec = vec / norm[:, None]
        vec[~(norm > 0.1 * scale)] = np.nan  # reference at the centre: no axis
        return vec

    # -- flags and queues ----------------------------------------------------

    def min_overlap(self, child: str, parent: str) -> float:
        for rule in (self._min_overlap, self._recorded_overlap):
            if isinstance(rule, Mapping):
                value = (rule.get(child) or {}).get(parent)
                if value is not None:
                    return float(value)
            elif rule is not None:
                return float(rule)
        return 0.5

    def flag_table(self, name: str) -> "pd.DataFrame":
        """Flagged objects of *name*: ``score`` and ``partner`` (-1: none),
        indexed by label, most suspicious first.

        Cached until the next decision: the panel asks several times per
        key press, and a poor segmentation can flag a large share of
        hundreds of thousands of objects.
        """
        if name not in self._flag_cache:
            rows = self._flag_rows(name)
            g = rows.groupby("label")
            table = (g["score"].max() + 0.1 * (g.size() - 1)).to_frame("score")
            table["partner"] = g["partner"].max()
            table = table.sort_index().sort_values(
                "score", ascending=False, kind="stable"
            )
            self._flag_cache[name] = (table, rows)
        return self._flag_cache[name][0]

    def reasons(self, name: str, label: int) -> list[str]:
        """Why object *label* of *name* is flagged (empty if it isn't)."""
        self.flag_table(name)
        rows = self._flag_cache[name][1]
        return rows.loc[rows["label"] == int(label), "reason"].tolist()

    def flags(self, name: str) -> list[Flag]:
        """Every flagged object of *name* with its reasons, most suspicious
        first. See :meth:`flag_table` for the fast form."""
        table = self.flag_table(name)
        rows = self._flag_cache[name][1]
        by_label = rows.groupby("label")["reason"].apply(list)
        return [
            Flag(
                int(label),
                by_label[label],
                float(r.score),
                None if r.partner < 0 else int(r.partner),
            )
            for label, r in zip(table.index, table.itertuples())
        ]

    def _flag_rows(self, name: str) -> "pd.DataFrame":
        """One row per (object, reason): label, reason, score, partner."""
        import pandas as pd

        eff = self.effective(name)
        ids = eff.index.to_numpy()
        parts = []

        def rule_(mask, reasons, scores, partners=None):
            labels = ids[mask]
            if labels.size:
                parts.append(
                    pd.DataFrame(
                        {
                            "label": labels.astype(np.int64),
                            "reason": list(reasons),
                            "score": np.broadcast_to(
                                np.asarray(scores, float), labels.shape
                            ),
                            "partner": (
                                -1
                                if partners is None
                                else np.asarray(partners, np.int64)
                            ),
                        }
                    )
                )

        for p in self.parents[name]:
            pid = eff[f"{p}_id"].to_numpy()
            ov = eff[f"{p}_overlap"].to_numpy(dtype=float)
            limit = self.min_overlap(name, p)
            nice = _nice(p)
            orphan = pid == 0
            by_distance = f"{p}_distance_um" in eff
            rule_(
                orphan,
                [
                    f"not inside or near any {nice}"
                    if by_distance
                    else f"not inside any {nice}"
                ]
                * int(orphan.sum()),
                3.0,
            )
            gap = (
                eff[f"{p}_distance_um"].to_numpy(dtype=float)
                if by_distance
                else np.zeros(len(eff))
            )
            near = (pid != 0) & (gap > 0)
            rule_(
                near,
                (
                    f"outside {nice} #{q}, {g:.2g} µm away"
                    for q, g in zip(pid[near], gap[near])
                ),
                1.5,
            )
            weak = (pid != 0) & (ov < limit) & ~(gap > 0)
            rule_(
                weak,
                (
                    f"only {v:.0%} inside {nice} #{q}"
                    for v, q in zip(ov[weak], pid[weak])
                ),
                2.0 + (limit - ov[weak]),
            )
        for child, (lo, hi) in self.expect.get(name, {}).items():
            if f"n_{child}" not in eff:
                continue
            n = eff[f"n_{child}"].to_numpy()
            off = (n < lo) | (n > hi)
            want = f"{lo}" if lo == hi else f"{lo}-{hi}"
            rule_(
                off,
                (f"{k} {_nice(child)} (expected {want})" for k in n[off]),
                2.0,
            )
        if name in self.position and "position" in eff:
            rule = self.position[name]
            unknown = (eff["position"] == "unknown").to_numpy()
            ref = str(rule.get("apical", "")).removeprefix("towards:")
            rule_(
                unknown,
                [
                    f"position unclear: its {_nice(rule['parent'])} has no "
                    f"{_nice(ref)} to orient it"
                ]
                * int(unknown.sum()),
                1.2,
            )
        if len(eff) >= 20:
            logv = np.log(eff["area_voxels"].to_numpy(dtype=float))
            med = np.median(logv)
            mad = 1.4826 * np.median(np.abs(logv - med))
            if mad > 0:
                z = (logv - med) / mad
                odd = np.abs(z) > 3.5
                rule_(
                    odd,
                    (
                        f"unusually large ({math.exp(lv - med):.1f}x the median)"
                        if zz > 0
                        else f"unusually small ({math.exp(lv - med):.2f}x the median)"
                        for zz, lv in zip(z[odd], logv[odd])
                    ),
                    1.0 + np.minimum(np.abs(z[odd]) / 10, 1.0),
                )
        pairs = self._seam_pairs(name, eff)
        if pairs:
            a, b, axes = (np.array(x) for x in zip(*pairs))
            for one, other in ((a, b), (b, a)):
                mask = np.isin(ids, one)
                order = pd.Series(other, index=one)
                order = order[~order.index.duplicated()]
                partners = order.reindex(ids[mask]).to_numpy()
                ax = pd.Series(axes, index=one)
                ax = ax[~ax.index.duplicated()].reindex(ids[mask]).to_numpy()
                rule_(
                    mask,
                    (
                        f"meets #{q} exactly at a tile seam ({x}): "
                        "one object split?"
                        for q, x in zip(partners, ax)
                    ),
                    2.5,
                    partners,
                )
        if not parts:
            return pd.DataFrame(
                {
                    "label": np.empty(0, np.int64),
                    "reason": [],
                    "score": np.empty(0),
                    "partner": np.empty(0, np.int64),
                }
            )
        return pd.concat(parts, ignore_index=True)

    def _seam_pairs(self, name: str, eff: "pd.DataFrame"):
        """Pairs of objects that meet face to face on a tile boundary.

        Two objects, one ending on the seam and the other starting there,
        whose footprints across the seam coincide: either two neighbours
        pressed together, or one object the tiling cut in two. Worth a look
        either way; from bounding boxes alone, so no voxels are read.
        """
        from scipy.spatial import cKDTree

        meta = self.meta[name]
        tile = meta.get("tile_shape")
        axes = meta.get("axes") or ""
        if not tile or len(eff) < 2 or not axes:
            return []
        shape = meta["labels"]["shape"]
        tile = list(tile)[-len(axes) :]
        tol = self.seam_tolerance
        lo = {ax: eff[f"bbox_min_{ax}"].to_numpy() for ax in axes}
        hi = {ax: eff[f"bbox_max_{ax}"].to_numpy() + 1 for ax in axes}
        ids = eff.index.to_numpy()
        pairs = []
        for k, ax in enumerate(axes):
            t = int(tile[k])
            if t <= 0 or t >= shape[k]:
                continue
            others = [a for a in axes if a != ax]
            end_s = np.round(hi[ax] / t) * t
            beg_s = np.round(lo[ax] / t) * t
            ok_m = (
                (np.abs(hi[ax] - end_s) <= tol)
                & (end_s > 0)
                & (end_s < shape[k])
            )
            ok_p = (
                (np.abs(lo[ax] - beg_s) <= tol)
                & (beg_s > 0)
                & (beg_s < shape[k])
            )
            for s in np.intersect1d(end_s[ok_m], beg_s[ok_p]):
                m = np.flatnonzero(ok_m & (end_s == s))
                p = np.flatnonzero(ok_p & (beg_s == s))
                if not m.size or not p.size:
                    continue

                def box(idx):
                    b0 = np.stack([lo[a][idx] for a in others], 1).astype(float)
                    b1 = np.stack([hi[a][idx] for a in others], 1).astype(float)
                    return b0, b1

                m0, m1 = box(m)
                p0, p1 = box(p)
                pc = (p0 + p1) / 2
                reach = np.linalg.norm(m1 - m0, axis=1) / 2 + (
                    np.linalg.norm(p1 - p0, axis=1).max() / 2
                )
                tree = cKDTree(pc)
                for i, js in enumerate(
                    tree.query_ball_point((m0 + m1) / 2, reach)
                ):
                    if not js:
                        continue
                    js = np.asarray(js)
                    inter = np.prod(
                        np.clip(
                            np.minimum(m1[i], p1[js])
                            - np.maximum(m0[i], p0[js]),
                            0,
                            None,
                        ),
                        axis=1,
                    )
                    small = np.minimum(
                        np.prod(m1[i] - m0[i]), np.prod(p1[js] - p0[js], axis=1)
                    )
                    exact = (hi[ax][m[i]] == s) | (lo[ax][p[js]] == s)
                    for j in js[(inter >= 0.5 * small) & exact]:
                        if ids[m[i]] != ids[p[j]]:
                            pairs.append((int(ids[m[i]]), int(ids[p[j]]), ax))
        return pairs

    def queue(self, name: str, mode: str = "flagged") -> list[int]:
        """Objects still to review, in review order.

        ``flagged``: flagged objects, most suspicious first. ``random``: a
        fixed random order over every object (the error estimate). ``all``:
        every object by id.
        """
        done = self.decisions[name]
        if mode == "flagged":
            return [
                int(x) for x in self.flag_table(name).index if x not in done
            ]
        if mode == "random":
            return [x for x in self.random_order(name) if x not in done]
        if mode == "all":
            return [int(x) for x in self.effective(name).index if x not in done]
        raise ValueError(f"queue must be one of {QUEUES}, got {mode!r}")

    def random_order(self, name: str) -> list[int]:
        ids = self.tables[name].index.to_numpy()
        rng = np.random.default_rng(self.seed[name])
        return [int(x) for x in ids[rng.permutation(ids.size)]]

    def summary(self, name: str) -> dict[str, Any]:
        """Counts of objects, flags and decisions, and the error estimate.

        The estimate uses the longest run of the random order that has been
        fully reviewed (from any queue: a decision is the truth about that
        object however it came up), so it stays unbiased.
        """
        dec = self.decisions[name]
        # A segmentation error: rejected, joined or re-parented. A position
        # correction is a classification fix, counted on its own.
        wrong = {
            k
            for k, d in dec.items()
            if d["action"] in ("wrong", "merge", "parent")
        }
        sample = 0
        errors = 0
        for label in self.random_order(name):
            if label not in dec:
                break
            sample += 1
            errors += label in wrong
        flagged = self.flag_table(name).index
        lo, hi = wilson_interval(errors, sample)
        return {
            "objects": int(len(self.tables[name])),
            "flagged": len(flagged),
            "flagged_open": sum(int(x) not in dec for x in flagged),
            "reviewed": len(dec),
            "ok": sum(d["action"] == "ok" for d in dec.values()),
            "wrong": sum(d["action"] == "wrong" for d in dec.values()),
            "fixed": sum(
                d["action"] in ("parent", "merge") for d in dec.values()
            ),
            "position_fixed": sum(
                d["action"] == "position" for d in dec.values()
            ),
            "sample": sample,
            "sample_errors": errors,
            "error_rate": errors / sample if sample else None,
            "error_ci": (lo, hi) if sample else None,
        }

    # -- outputs -------------------------------------------------------------

    def export(self, out_dir: Union[str, Path], fmt: str = "csv") -> list[Path]:
        """Write every corrected table to *out_dir*, one file per label image.

        ``csv`` files load straight into napari-chunked-regionprops
        ("Reload previous results"). ``xlsx`` falls back to csv for a table
        longer than Excel allows. ``parquet`` needs pyarrow.
        """
        out = Path(out_dir)
        out.mkdir(parents=True, exist_ok=True)
        written = []
        for name in self.names:
            df = self.effective(name)
            kind = fmt
            if fmt == "xlsx" and len(df) + 1 > EXCEL_MAX_ROWS:
                logger.warning(
                    "%s has %d objects, more than an Excel sheet holds; "
                    "writing csv instead",
                    name,
                    len(df),
                )
                kind = "csv"
            path = out / f"{name}.{kind}"
            if kind == "csv":
                df.to_csv(path)
            elif kind == "xlsx":
                df.to_excel(path, sheet_name=name[:31])
            elif kind == "parquet":
                df.to_parquet(path)
            else:
                raise ValueError(
                    f"format must be csv, xlsx or parquet: {fmt!r}"
                )
            written.append(path)
        return written

    def relation_workbook(
        self, child: str, parent: str, path: Union[str, Path]
    ) -> Path:
        """Write the *child* -> *parent* workbook from the corrected tables.

        Sheet *child*: each object, its parent, overlap and review status.
        Sheet *parent*: each parent object and how many children it holds.
        Written as two csv files (``<stem>_<name>.csv``) when a sheet would
        exceed Excel's row limit.
        """
        path = Path(path)
        ce = self.effective(child)
        pe = self.effective(parent)
        a = ce[
            [f"{parent}_id", f"{parent}_overlap_voxels", f"{parent}_overlap"]
        ]
        a = a.rename(
            columns={
                f"{parent}_id": f"{parent}_id",
                f"{parent}_overlap_voxels": "overlap_voxels",
                f"{parent}_overlap": "overlap_fraction",
            }
        ).copy()
        a[f"{parent}_id"] = a[f"{parent}_id"].where(a[f"{parent}_id"] != 0)
        if (
            "position" in ce
            and self.position.get(child, {}).get("parent") == parent
        ):
            a["position"] = ce["position"]
        a["qc"] = ce["qc"]
        a.index.name = f"{child}_id"
        grouped = ce[ce[f"{parent}_id"] != 0].groupby(f"{parent}_id")
        b = pe[[]].copy()
        b[f"{child}_count"] = (
            grouped.size().reindex(pe.index).fillna(0).astype(np.int64)
        )
        b["total_overlap_voxels"] = (
            grouped[f"{parent}_overlap_voxels"]
            .sum()
            .reindex(pe.index)
            .fillna(0)
            .astype(np.int64)
        )
        for cls in POSITIONS:
            col = f"n_{child}_{cls}"
            if col in pe:
                b[f"{child}_{cls}"] = pe[col]
        b["qc"] = pe["qc"]
        b.index.name = f"{parent}_id"
        if max(len(a), len(b)) + 1 > EXCEL_MAX_ROWS:
            logger.warning(
                "%s -> %s exceeds Excel's row limit; writing csv files",
                child,
                parent,
            )
            a.to_csv(path.with_name(f"{path.stem}_{child}.csv"))
            b.to_csv(path.with_name(f"{path.stem}_{parent}.csv"))
            return path.with_name(f"{path.stem}_{child}.csv")
        import pandas as pd

        with pd.ExcelWriter(path) as xl:
            a.to_excel(xl, sheet_name=child[:31])
            b.to_excel(xl, sheet_name=parent[:31])
        return path

    def write_reviewed_labels(
        self, name: str, *, suffix: str = "_reviewed"
    ) -> str:
        """Write ``labels/<name><suffix>``: the labels with the decisions
        applied to the voxels (rejected objects erased, merged objects one
        id), plus its corrected table.

        The corrected tables and workbooks never need this; it is for figures
        and for tools that read label images only.
        """
        import dask.array as da

        from ._provenance import PROVENANCE_KEY
        from ._tables import _level0, write_table
        from .plugins.ome_zarr import write_labels

        group = zarr.open_group(_label_group(self.store, name), mode="r")
        level0 = da.from_zarr(_level0(group))
        roots = self.roots(name)
        top = int(max(self.tables[name].index.max(), max(roots, default=0)))
        lut = np.arange(top + 1, dtype=level0.dtype)
        for label, root in roots.items():
            lut[label] = root
        fixed = level0.map_blocks(lambda b: lut[b], dtype=level0.dtype)
        prov = dict(group.attrs).get(PROVENANCE_KEY) or {}
        settings = prov.get("settings") or {}
        record = dict(prov)
        record["review"] = {
            "from": name,
            "decisions": len(self.decisions[name]),
        }
        out = f"{name}{suffix}"
        write_labels(
            self.store,
            fixed,
            name=out,
            overwrite=True,
            level=int(settings.get("level") or 0),
            progress=False,
            provenance=record,
        )
        eff = self.effective(name).drop(columns=["qc"])
        cols = {"label": eff.index.to_numpy()}
        cols.update({c: eff[c].to_numpy() for c in eff.columns})
        meta = {
            k: v
            for k, v in self.meta[name].items()
            if k not in ("labels", "columns", "parents")
        }
        write_table(
            _label_group(self.store, out),
            cols,
            attrs={**meta, "reviewed_from": name},
        )
        return out

decide(name: str, label: int, action: str, *, parent: str | None = None, parent_id: int | None = None, into: int | None = None, position: str | None = None, queue: str = 'flagged', reviewer: str | None = None) -> None

Record a decision about object label of label image name.

Parameters:

Name Type Description Default
action ('ok', 'wrong', 'parent', 'merge')

ok: correct as it is. wrong: not a real object (dropped from the corrected tables). parent: belongs to parent_id of label image parent (0 = to none). merge: the same object as into. position: its position class is position (one of :data:POSITIONS), whatever was computed.

"ok"
queue str

The queue it was reviewed from; "random" decisions feed the error estimate.

'flagged'
Source code in src/patchworks/_review.py
def decide(
    self,
    name: str,
    label: int,
    action: str,
    *,
    parent: str | None = None,
    parent_id: int | None = None,
    into: int | None = None,
    position: str | None = None,
    queue: str = "flagged",
    reviewer: str | None = None,
) -> None:
    """Record a decision about object *label* of label image *name*.

    Parameters
    ----------
    action : {"ok", "wrong", "parent", "merge"}
        ``ok``: correct as it is. ``wrong``: not a real object (dropped
        from the corrected tables). ``parent``: belongs to *parent_id*
        of label image *parent* (0 = to none). ``merge``: the same
        object as *into*. ``position``: its position class is
        *position* (one of :data:`POSITIONS`), whatever was computed.
    queue : str
        The queue it was reviewed from; ``"random"`` decisions feed the
        error estimate.
    """
    if name not in self.tables:
        raise KeyError(f"no table for labels/{name}")
    label = int(label)
    if label not in self.tables[name].index:
        raise KeyError(f"{name} has no object {label}")
    if action not in ACTIONS:
        raise ValueError(f"action must be one of {ACTIONS}, got {action!r}")
    entry: dict[str, Any] = {
        "action": action,
        "queue": queue,
        "time": _dt.datetime.now(_dt.timezone.utc).isoformat(
            timespec="seconds"
        ),
    }
    if reviewer:
        entry["by"] = reviewer
    if action == "parent":
        if parent not in self.parents[name]:
            raise ValueError(
                f"{name} has no parent image {parent!r} "
                f"(has {self.parents[name]})"
            )
        pid = int(parent_id or 0)
        if pid and pid not in self.tables[parent].index:
            raise KeyError(f"{parent} has no object {pid}")
        entry.update(parent=parent, parent_id=pid)
        previous = self.decisions[name].get(label, {})
        if previous.get("action") == "parent":
            entry["parents"] = {**previous.get("parents", {}), parent: pid}
        else:
            entry["parents"] = {parent: pid}
    elif action == "position":
        if name not in self.position:
            raise ValueError(f"{name} has no position rule")
        if position not in POSITIONS:
            raise ValueError(f"position must be one of {POSITIONS}")
        entry["position"] = position
    elif action == "merge":
        into = int(into or 0)
        if into == label or into not in self.tables[name].index:
            raise KeyError(f"cannot merge {name} {label} into {into}")
        entry["into"] = into
    self.decisions[name][label] = entry
    self._save(name)

relations() -> list[tuple[str, str]]

Every (child, parent) pair the tables relate.

Source code in src/patchworks/_review.py
def relations(self) -> list[tuple[str, str]]:
    """Every (child, parent) pair the tables relate."""
    return [(c, p) for c in self.names for p in self.parents[c]]

undo(name: str, label: int) -> None

Forget the decision about label (it goes back to unreviewed).

Source code in src/patchworks/_review.py
def undo(self, name: str, label: int) -> None:
    """Forget the decision about *label* (it goes back to unreviewed)."""
    if self.decisions[name].pop(int(label), None) is not None:
        self._save(name)

roots(name: str) -> dict[int, int]

Where each decided object ends up: 0 if rejected, the object it was merged into (followed to the end) otherwise. Undecided objects are absent (they map to themselves).

Source code in src/patchworks/_review.py
def roots(self, name: str) -> dict[int, int]:
    """Where each decided object ends up: 0 if rejected, the object it
    was merged into (followed to the end) otherwise. Undecided objects
    are absent (they map to themselves)."""
    dec = self.decisions[name]
    out: dict[int, int] = {}
    for label in dec:
        seen, cur = {label}, label
        while True:
            d = dec.get(cur)
            if d is None or d["action"] in ("ok", "parent", "position"):
                break
            if d["action"] == "wrong":
                cur = 0
                break
            cur = int(d["into"])
            if cur in seen:  # a merge cycle: keep the smallest id
                cur = min(seen)
                break
            seen.add(cur)
        if cur != label:
            out[label] = cur
    return out

effective(name: str) -> 'pd.DataFrame'

name's table with every decision applied.

Rejected objects are dropped; merged objects become one row (sizes added, centroids, spreads and intensities combined exactly, boxes joined); parent columns follow reassignments and the parents' own merges; qc says ok, fixed or "" (not reviewed). Derived columns: n_<child> (children counted), length_um, elongation, axis_<a> (shape, see :func:~patchworks._tables.shape_columns) and, for label images with a position rule, position and its measures (see :meth:positions).

Source code in src/patchworks/_review.py
def effective(self, name: str) -> "pd.DataFrame":
    """*name*'s table with every decision applied.

    Rejected objects are dropped; merged objects become one row (sizes
    added, centroids, spreads and intensities combined exactly, boxes
    joined); parent columns follow reassignments and the parents' own
    merges; ``qc`` says ``ok``, ``fixed`` or ``""`` (not reviewed).
    Derived columns: ``n_<child>`` (children counted), ``length_um``,
    ``elongation``, ``axis_<a>`` (shape, see
    :func:`~patchworks._tables.shape_columns`) and, for label images
    with a position rule, ``position`` and its measures (see
    :meth:`positions`).
    """
    if name in self._cache:
        return self._cache[name]
    from ._tables import shape_columns

    df = self._corrected(name).copy()
    for child in self.children[name]:
        ce = self._corrected(child)
        counts = ce[f"{name}_id"].value_counts()
        df[f"n_{child}"] = (
            counts.reindex(df.index).fillna(0).astype(np.int64)
        )
    meta = self.meta[name]
    for col, values in shape_columns(
        df, meta.get("axes") or "", meta.get("pixel_size")
    ).items():
        df[col] = values
    if name in self.position:
        for col, values in self.positions(name, df).items():
            df[col] = values
    for child in self.children[name]:
        if (
            child in self.position
            and self.position[child]["parent"] == name
        ):
            pos = self.effective(child)
            table = (
                pos[pos[f"{name}_id"] != 0]
                .groupby([f"{name}_id", "position"])
                .size()
                .unstack(fill_value=0)
            )
            for cls in POSITIONS:
                counts = table[cls] if cls in table else None
                df[f"n_{child}_{cls}"] = (
                    counts.reindex(df.index).fillna(0).astype(np.int64)
                    if counts is not None
                    else 0
                )
    self._cache[name] = df
    return df

positions(name: str, df: 'pd.DataFrame | None' = None) -> dict[str, Any]

Where each object of name sits in its parent.

Per parent object, an apical axis: a fixed direction ("+z": apical is up the z axis), or pointing away from the parent's children of another label image ("nuclei_labels": away from the nucleus, for epithelia whose nuclei sit basally), or towards them ("towards:<name>"). Each object's base is its end nearer the parent's centre (a cilium grows out from its base).

The class is the parent surface the base is nearest to -- top (apical), bottom (basal) or side wall (lateral) -- each depth measured relative to the parent's size in that direction; central when deeper than central_depth (default 0.5, i.e. half-way) from all three. The parent's shape is taken from its moments (an equivalent cylinder), so this is a classification, not a surface distance.

Returns columns position (the class; outside with no parent; unknown when the axis cannot be told, e.g. no nucleus), position_axial (-1 basal .. +1 apical, in half-heights of the parent along its axis), position_radial (0 on the axis .. 1 at the side) and angle_to_axis_deg (0: along the apical axis, 90: across it). A reviewer's position decision overrides the class.

Source code in src/patchworks/_review.py
def positions(
    self, name: str, df: "pd.DataFrame | None" = None
) -> dict[str, Any]:
    """Where each object of *name* sits in its parent.

    Per parent object, an apical axis: a fixed direction (``"+z"``:
    apical is up the z axis), or pointing away from the parent's
    children of another label image (``"nuclei_labels"``: away from
    the nucleus, for epithelia whose nuclei sit basally), or towards
    them (``"towards:<name>"``). Each object's **base** is its end
    nearer the parent's centre (a cilium grows out from its base).

    The class is the parent surface the base is nearest to -- top
    (``apical``), bottom (``basal``) or side wall (``lateral``) --
    each depth measured relative to the parent's size in that
    direction; ``central`` when deeper than ``central_depth`` (default
    0.5, i.e. half-way) from all three. The parent's shape is taken
    from its moments (an equivalent cylinder), so this is a
    classification, not a surface distance.

    Returns columns ``position`` (the class; ``outside`` with no
    parent; ``unknown`` when the axis cannot be told, e.g. no nucleus),
    ``position_axial`` (-1 basal .. +1 apical, in half-heights of the
    parent along its axis), ``position_radial`` (0 on the axis .. 1 at
    the side) and ``angle_to_axis_deg`` (0: along the apical axis, 90:
    across it). A reviewer's ``position`` decision overrides the class.
    """
    from ._tables import physical_centroids, physical_cov

    rule = self.position[name]
    parent = rule["parent"]
    df = self._corrected(name) if df is None else df
    n = len(df)
    out: dict[str, Any] = {
        "position": np.full(n, "unknown", dtype=object),
        "position_axial": np.full(n, np.nan),
        "position_radial": np.full(n, np.nan),
        "angle_to_axis_deg": np.full(n, np.nan),
    }
    if parent not in self.parents[name] or not n:
        return out
    axes = self.meta[name].get("axes") or ""
    size = self.meta[name].get("pixel_size")
    pe = self._corrected(parent)
    pid = df[f"{parent}_id"].to_numpy(dtype=np.int64)
    spread = [f"cov_{a}{a}" for a in axes]
    missing = [
        t
        for t, d in ((name, df), (parent, pe))
        if not set(spread) <= set(d)
    ]
    if missing:
        logger.warning(
            "no spread (cov_*) columns in the table(s) of %s -- written "
            "by an older patchworks; recompute them (patchworks tables "
            "%s) to classify positions",
            ", ".join(missing),
            self.store,
        )
        out["position"][pid == 0] = "outside"
        return out
    out["position"][pid == 0] = "outside"
    inside = np.flatnonzero(np.isin(pid, pe.index.to_numpy()))
    if not inside.size:
        return out
    prow = pe.index.get_indexer(pid[inside])
    centre = physical_centroids(pe, axes, size)[prow]
    pcov = physical_cov(pe, axes, size)[prow]
    axis = self._apical_axes(name, rule, pe, axes, size)[prow]

    c = physical_centroids(df, axes, size)[inside]
    w, v = np.linalg.eigh(physical_cov(df, axes, size)[inside])
    main = v[:, :, -1]
    half = np.sqrt(12 * np.clip(w[:, -1], 0, None)) / 2
    ends = np.stack([c + half[:, None] * main, c - half[:, None] * main])
    near = np.argmin(np.linalg.norm(ends - centre, axis=2), axis=0)
    base = ends[near, np.arange(inside.size)]

    rel = base - centre
    along = np.einsum("ij,ij->i", rel, axis)
    var_a = np.einsum("ij,ijk,ik->i", axis, pcov, axis)
    axial = along / np.sqrt(3 * np.clip(var_a, 1e-12, None))
    perp = rel - along[:, None] * axis
    across = np.sqrt(
        2 * np.clip(np.trace(pcov, axis1=1, axis2=2) - var_a, 1e-12, None)
    )
    radial = np.linalg.norm(perp, axis=1) / across
    angle = np.degrees(
        np.arccos(np.clip(np.abs(np.einsum("ij,ij->i", main, axis)), 0, 1))
    )
    # The surface the base is nearest to, each depth relative to the
    # parent's size in that direction: top, bottom or side wall. Deeper
    # than `central_depth` from all three: central.
    depth = np.stack([1 - axial, 1 + axial, 1 - radial], 1)
    cls = np.array(["apical", "basal", "lateral"], dtype=object)[
        np.argmin(depth, axis=1)
    ]
    cls[depth.min(axis=1) > float(rule.get("central_depth", 0.5))] = (
        "central"
    )
    known = np.isfinite(axis).all(axis=1)
    cls[~known] = "unknown"
    out["position"][inside] = cls
    out["position_axial"][inside] = np.where(known, axial, np.nan)
    out["position_radial"][inside] = np.where(known, radial, np.nan)
    out["angle_to_axis_deg"][inside] = np.where(known, angle, np.nan)
    for label, d in self.decisions[name].items():
        if d["action"] == "position" and label in df.index:
            out["position"][df.index.get_loc(label)] = d["position"]
    return out

flag_table(name: str) -> 'pd.DataFrame'

Flagged objects of name: score and partner (-1: none), indexed by label, most suspicious first.

Cached until the next decision: the panel asks several times per key press, and a poor segmentation can flag a large share of hundreds of thousands of objects.

Source code in src/patchworks/_review.py
def flag_table(self, name: str) -> "pd.DataFrame":
    """Flagged objects of *name*: ``score`` and ``partner`` (-1: none),
    indexed by label, most suspicious first.

    Cached until the next decision: the panel asks several times per
    key press, and a poor segmentation can flag a large share of
    hundreds of thousands of objects.
    """
    if name not in self._flag_cache:
        rows = self._flag_rows(name)
        g = rows.groupby("label")
        table = (g["score"].max() + 0.1 * (g.size() - 1)).to_frame("score")
        table["partner"] = g["partner"].max()
        table = table.sort_index().sort_values(
            "score", ascending=False, kind="stable"
        )
        self._flag_cache[name] = (table, rows)
    return self._flag_cache[name][0]

reasons(name: str, label: int) -> list[str]

Why object label of name is flagged (empty if it isn't).

Source code in src/patchworks/_review.py
def reasons(self, name: str, label: int) -> list[str]:
    """Why object *label* of *name* is flagged (empty if it isn't)."""
    self.flag_table(name)
    rows = self._flag_cache[name][1]
    return rows.loc[rows["label"] == int(label), "reason"].tolist()

flags(name: str) -> list[Flag]

Every flagged object of name with its reasons, most suspicious first. See :meth:flag_table for the fast form.

Source code in src/patchworks/_review.py
def flags(self, name: str) -> list[Flag]:
    """Every flagged object of *name* with its reasons, most suspicious
    first. See :meth:`flag_table` for the fast form."""
    table = self.flag_table(name)
    rows = self._flag_cache[name][1]
    by_label = rows.groupby("label")["reason"].apply(list)
    return [
        Flag(
            int(label),
            by_label[label],
            float(r.score),
            None if r.partner < 0 else int(r.partner),
        )
        for label, r in zip(table.index, table.itertuples())
    ]

queue(name: str, mode: str = 'flagged') -> list[int]

Objects still to review, in review order.

flagged: flagged objects, most suspicious first. random: a fixed random order over every object (the error estimate). all: every object by id.

Source code in src/patchworks/_review.py
def queue(self, name: str, mode: str = "flagged") -> list[int]:
    """Objects still to review, in review order.

    ``flagged``: flagged objects, most suspicious first. ``random``: a
    fixed random order over every object (the error estimate). ``all``:
    every object by id.
    """
    done = self.decisions[name]
    if mode == "flagged":
        return [
            int(x) for x in self.flag_table(name).index if x not in done
        ]
    if mode == "random":
        return [x for x in self.random_order(name) if x not in done]
    if mode == "all":
        return [int(x) for x in self.effective(name).index if x not in done]
    raise ValueError(f"queue must be one of {QUEUES}, got {mode!r}")

summary(name: str) -> dict[str, Any]

Counts of objects, flags and decisions, and the error estimate.

The estimate uses the longest run of the random order that has been fully reviewed (from any queue: a decision is the truth about that object however it came up), so it stays unbiased.

Source code in src/patchworks/_review.py
def summary(self, name: str) -> dict[str, Any]:
    """Counts of objects, flags and decisions, and the error estimate.

    The estimate uses the longest run of the random order that has been
    fully reviewed (from any queue: a decision is the truth about that
    object however it came up), so it stays unbiased.
    """
    dec = self.decisions[name]
    # A segmentation error: rejected, joined or re-parented. A position
    # correction is a classification fix, counted on its own.
    wrong = {
        k
        for k, d in dec.items()
        if d["action"] in ("wrong", "merge", "parent")
    }
    sample = 0
    errors = 0
    for label in self.random_order(name):
        if label not in dec:
            break
        sample += 1
        errors += label in wrong
    flagged = self.flag_table(name).index
    lo, hi = wilson_interval(errors, sample)
    return {
        "objects": int(len(self.tables[name])),
        "flagged": len(flagged),
        "flagged_open": sum(int(x) not in dec for x in flagged),
        "reviewed": len(dec),
        "ok": sum(d["action"] == "ok" for d in dec.values()),
        "wrong": sum(d["action"] == "wrong" for d in dec.values()),
        "fixed": sum(
            d["action"] in ("parent", "merge") for d in dec.values()
        ),
        "position_fixed": sum(
            d["action"] == "position" for d in dec.values()
        ),
        "sample": sample,
        "sample_errors": errors,
        "error_rate": errors / sample if sample else None,
        "error_ci": (lo, hi) if sample else None,
    }

export(out_dir: Union[str, Path], fmt: str = 'csv') -> list[Path]

Write every corrected table to out_dir, one file per label image.

csv files load straight into napari-chunked-regionprops ("Reload previous results"). xlsx falls back to csv for a table longer than Excel allows. parquet needs pyarrow.

Source code in src/patchworks/_review.py
def export(self, out_dir: Union[str, Path], fmt: str = "csv") -> list[Path]:
    """Write every corrected table to *out_dir*, one file per label image.

    ``csv`` files load straight into napari-chunked-regionprops
    ("Reload previous results"). ``xlsx`` falls back to csv for a table
    longer than Excel allows. ``parquet`` needs pyarrow.
    """
    out = Path(out_dir)
    out.mkdir(parents=True, exist_ok=True)
    written = []
    for name in self.names:
        df = self.effective(name)
        kind = fmt
        if fmt == "xlsx" and len(df) + 1 > EXCEL_MAX_ROWS:
            logger.warning(
                "%s has %d objects, more than an Excel sheet holds; "
                "writing csv instead",
                name,
                len(df),
            )
            kind = "csv"
        path = out / f"{name}.{kind}"
        if kind == "csv":
            df.to_csv(path)
        elif kind == "xlsx":
            df.to_excel(path, sheet_name=name[:31])
        elif kind == "parquet":
            df.to_parquet(path)
        else:
            raise ValueError(
                f"format must be csv, xlsx or parquet: {fmt!r}"
            )
        written.append(path)
    return written

relation_workbook(child: str, parent: str, path: Union[str, Path]) -> Path

Write the child -> parent workbook from the corrected tables.

Sheet child: each object, its parent, overlap and review status. Sheet parent: each parent object and how many children it holds. Written as two csv files (<stem>_<name>.csv) when a sheet would exceed Excel's row limit.

Source code in src/patchworks/_review.py
def relation_workbook(
    self, child: str, parent: str, path: Union[str, Path]
) -> Path:
    """Write the *child* -> *parent* workbook from the corrected tables.

    Sheet *child*: each object, its parent, overlap and review status.
    Sheet *parent*: each parent object and how many children it holds.
    Written as two csv files (``<stem>_<name>.csv``) when a sheet would
    exceed Excel's row limit.
    """
    path = Path(path)
    ce = self.effective(child)
    pe = self.effective(parent)
    a = ce[
        [f"{parent}_id", f"{parent}_overlap_voxels", f"{parent}_overlap"]
    ]
    a = a.rename(
        columns={
            f"{parent}_id": f"{parent}_id",
            f"{parent}_overlap_voxels": "overlap_voxels",
            f"{parent}_overlap": "overlap_fraction",
        }
    ).copy()
    a[f"{parent}_id"] = a[f"{parent}_id"].where(a[f"{parent}_id"] != 0)
    if (
        "position" in ce
        and self.position.get(child, {}).get("parent") == parent
    ):
        a["position"] = ce["position"]
    a["qc"] = ce["qc"]
    a.index.name = f"{child}_id"
    grouped = ce[ce[f"{parent}_id"] != 0].groupby(f"{parent}_id")
    b = pe[[]].copy()
    b[f"{child}_count"] = (
        grouped.size().reindex(pe.index).fillna(0).astype(np.int64)
    )
    b["total_overlap_voxels"] = (
        grouped[f"{parent}_overlap_voxels"]
        .sum()
        .reindex(pe.index)
        .fillna(0)
        .astype(np.int64)
    )
    for cls in POSITIONS:
        col = f"n_{child}_{cls}"
        if col in pe:
            b[f"{child}_{cls}"] = pe[col]
    b["qc"] = pe["qc"]
    b.index.name = f"{parent}_id"
    if max(len(a), len(b)) + 1 > EXCEL_MAX_ROWS:
        logger.warning(
            "%s -> %s exceeds Excel's row limit; writing csv files",
            child,
            parent,
        )
        a.to_csv(path.with_name(f"{path.stem}_{child}.csv"))
        b.to_csv(path.with_name(f"{path.stem}_{parent}.csv"))
        return path.with_name(f"{path.stem}_{child}.csv")
    import pandas as pd

    with pd.ExcelWriter(path) as xl:
        a.to_excel(xl, sheet_name=child[:31])
        b.to_excel(xl, sheet_name=parent[:31])
    return path

write_reviewed_labels(name: str, *, suffix: str = '_reviewed') -> str

Write labels/<name><suffix>: the labels with the decisions applied to the voxels (rejected objects erased, merged objects one id), plus its corrected table.

The corrected tables and workbooks never need this; it is for figures and for tools that read label images only.

Source code in src/patchworks/_review.py
def write_reviewed_labels(
    self, name: str, *, suffix: str = "_reviewed"
) -> str:
    """Write ``labels/<name><suffix>``: the labels with the decisions
    applied to the voxels (rejected objects erased, merged objects one
    id), plus its corrected table.

    The corrected tables and workbooks never need this; it is for figures
    and for tools that read label images only.
    """
    import dask.array as da

    from ._provenance import PROVENANCE_KEY
    from ._tables import _level0, write_table
    from .plugins.ome_zarr import write_labels

    group = zarr.open_group(_label_group(self.store, name), mode="r")
    level0 = da.from_zarr(_level0(group))
    roots = self.roots(name)
    top = int(max(self.tables[name].index.max(), max(roots, default=0)))
    lut = np.arange(top + 1, dtype=level0.dtype)
    for label, root in roots.items():
        lut[label] = root
    fixed = level0.map_blocks(lambda b: lut[b], dtype=level0.dtype)
    prov = dict(group.attrs).get(PROVENANCE_KEY) or {}
    settings = prov.get("settings") or {}
    record = dict(prov)
    record["review"] = {
        "from": name,
        "decisions": len(self.decisions[name]),
    }
    out = f"{name}{suffix}"
    write_labels(
        self.store,
        fixed,
        name=out,
        overwrite=True,
        level=int(settings.get("level") or 0),
        progress=False,
        provenance=record,
    )
    eff = self.effective(name).drop(columns=["qc"])
    cols = {"label": eff.index.to_numpy()}
    cols.update({c: eff[c].to_numpy() for c in eff.columns})
    meta = {
        k: v
        for k, v in self.meta[name].items()
        if k not in ("labels", "columns", "parents")
    }
    write_table(
        _label_group(self.store, out),
        cols,
        attrs={**meta, "reviewed_from": name},
    )
    return out

patchworks.plugins.review.review_in_napari(store: Union[str, Path], *, expect: dict | None = None, min_overlap: float | None = None, position: dict | None = None, show: bool = True, viewer: Any = None)

Open store in napari with the review panel docked.

Parameters:

Name Type Description Default
store str or Path

OME-ZARR image store whose label images carry object tables (the workflow writes them; patchworks tables adds them).

required
expect dict | None

Review rules, see :class:patchworks._review.Review.

None
min_overlap dict | None

Review rules, see :class:patchworks._review.Review.

None
position dict | None

Review rules, see :class:patchworks._review.Review.

None
show bool

Start the napari event loop (blocking).

True
viewer Viewer

Reuse a viewer that already shows the store's layers.

None

Returns:

Type Description
tuple

(viewer, widget).

Source code in src/patchworks/plugins/review.py
def review_in_napari(
    store: Union[str, Path],
    *,
    expect: dict | None = None,
    min_overlap: float | None = None,
    position: dict | None = None,
    show: bool = True,
    viewer: Any = None,
):
    """Open *store* in napari with the review panel docked.

    Parameters
    ----------
    store : str or Path
        OME-ZARR image store whose label images carry object tables
        (the workflow writes them; ``patchworks tables`` adds them).
    expect, min_overlap, position
        Review rules, see :class:`patchworks._review.Review`.
    show : bool
        Start the napari event loop (blocking).
    viewer : napari.Viewer, optional
        Reuse a viewer that already shows the store's layers.

    Returns
    -------
    tuple
        ``(viewer, widget)``.
    """
    from .napari import _require_napari, view_in_napari

    napari = _require_napari()
    rv = Review(
        store, expect=expect, min_overlap=min_overlap, position=position
    )
    if not rv.names:
        raise ValueError(
            f"no label image in {store} has an object table yet: run "
            f"`patchworks tables {store}` first"
            + (f" (stale: {', '.join(rv.stale)})" if rv.stale else "")
        )
    if viewer is None:
        viewer = view_in_napari(store, show=False)
    widget = ReviewWidget(viewer, rv)
    viewer.window.add_dock_widget(widget, name="Review", area="right")
    if show:
        napari.run()
    return viewer, widget