diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-BagOfChannels-v3.sh b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-BagOfChannels-v3.sh similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-BagOfChannels-v3.sh rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-BagOfChannels-v3.sh diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-BagOfChannels-v3.yml b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-BagOfChannels-v3.yml similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-BagOfChannels-v3.yml rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-BagOfChannels-v3.yml diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-classical.sh b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-classical.sh similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-classical.sh rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-classical.sh diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-classical.yml b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-classical.yml similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-classical.yml rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-classical.yml diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker-192.sh b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker-192.sh similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker-192.sh rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker-192.sh diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker-192.yml b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker-192.yml similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker-192.yml rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker-192.yml diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker-A40.sh b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker-A40.sh similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker-A40.sh rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker-A40.sh diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker-A40.yml b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker-A40.yml similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker-A40.yml rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker-A40.yml diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker.sh b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker.sh similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker.sh rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker.sh diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker.yml b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker.yml similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels-single-marker.yml rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-single-marker.yml diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels.sh b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels.sh similarity index 97% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels.sh rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels.sh index a74b670e9..c139321e6 100644 --- a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels.sh +++ b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels.sh @@ -16,6 +16,8 @@ #SBATCH --cpus-per-task=15 #SBATCH --mem-per-cpu=8G #SBATCH --time=3-00:00:00 +#SBATCH --requeue +#SBATCH --signal=B:USR1@300 # ── Run identity ────────────────────────────────────────────────────── # Warm-started from prior mixed-markers run (s1f8kgtp/last.ckpt, Apr 22) diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels.yml b/applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels.yml similarity index 100% rename from applications/dynaclr/configs/training/DynaCLR-2D/DynaCLR-2D-MIP-BagOfChannels.yml rename to applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels.yml diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain-classical.sh b/applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain-classical.sh new file mode 100644 index 000000000..e974ef4f9 --- /dev/null +++ b/applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain-classical.sh @@ -0,0 +1,37 @@ +#!/bin/bash +# DynaCLR-2D-MIP-pretrain CLASSICAL (SimCLR-style) variant — Reef/Kelp smoke test. +# Adapted from +# applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels-classical.sh +# per docs/clusters/reef.md: named GPU partition instead of --constraint, +# explicit --qos (Reef requires one). Uses the requeing_slurm branch's +# checkpoint+requeue support (train.sh always passes --slurm_auto_requeue; +# it only attaches SLURMEnvironment(auto_requeue=True) when actually running +# under SLURM). --requeue/--signal are harmless at --qos dev (non-preemptible) +# and keep this script ready to resubmit at --qos mid/low unchanged. +# +# sbatch applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain-classical.sh + +#SBATCH --job-name=dynaclr_2d_pretrain_smoke +#SBATCH --nodes=1 +#SBATCH --ntasks-per-node=2 +#SBATCH --gpus=2 +#SBATCH --partition=h100-reserved +#SBATCH --qos=dev +#SBATCH --cpus-per-task=15 +#SBATCH --mem-per-cpu=8G +#SBATCH --time=1-00:00:00 +#SBATCH --requeue +#SBATCH --signal=B:USR1@300 + +export WORKSPACE_DIR="/mnt/main0/home/eduardo.hirata/repos/VisCy" +export MODEL_ROOT="/bio/projects/compimaging/models" +export PROJECT="DynaCLR-2D-MIP-pretrain" +export RUN_NAME="2d-mip-classical-ntxent-t0p2-lr2e5-bs256-192to160-zext11-single-marker-reef-smoke" +export CONFIGS="applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain.yml applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain-classical.yml" + +# Smoke test: cap epochs/batches so we quickly get a checkpoint to validate +# against (single dirpath, no duplication) and to test SLURM preemption +# (auto-requeue + resume). Drop this override once the pipeline is validated. +export EXTRA_ARGS="--trainer.max_epochs=30 --trainer.limit_train_batches=5 --trainer.limit_val_batches=3" + +source "${WORKSPACE_DIR}/applications/dynaclr/configs/training/slurm/train.sh" diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain-classical.yml b/applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain-classical.yml new file mode 100644 index 000000000..6bf40d483 --- /dev/null +++ b/applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain-classical.yml @@ -0,0 +1,36 @@ +# Override: SimCLR-style "classical" DynaCLR-2D-MIP-pretrain. +# Anchor and positive are the same crop; augmentation creates two views. +# Same mock parquet as the base leaf; only the positive-sampling strategy +# and batching differ from the (not-yet-written) temporal-positive variant. + +data: + init_args: + cell_index_path: /bio/projects/compimaging/models/collections/DynaCLR-2D-MIP-pretrain-mock.parquet + # Self-augmented positives (SimCLR-style). Fully self-supervised: + # no lineage/track lookup, no tau scheduling. Augmentation pipeline + # provides view diversity. Single-marker batches still supply + # same-marker different-cell negatives, preserving discrimination + # signal at the marker/perturbation level. + positive_cell_source: self + positive_match_columns: null + positive_channel_source: same + # Single-marker batches (OPS strategy) — every batch is one marker, + # forcing the model to learn cellular features instead of channel + # shortcuts. + batch_group_by: marker + # Within a marker's draw, balance across the experiments containing + # that marker. + stratify_by: experiment + # Marker-uniform weights, restricted to the 5 markers actually present + # in DynaCLR-2D-MIP-pretrain-mock.parquet (G3BP1/TOMM20/SEC61B/ + # viral_sensor/Phase3D from the 5 mock 2026_* datasets). A marker key + # not present in the data is silently ignored; a marker present but + # missing from this dict gets sampling weight 0 (see + # viscy_data.sampler.FlexibleBatchSampler._precompute_groups) — so + # this list must be kept in sync with the collection's marker set. + group_weights: + Phase3D: 1.0 + G3BP1: 1.0 + SEC61B: 1.0 + TOMM20: 1.0 + viral_sensor: 1.0 diff --git a/applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain.yml b/applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain.yml new file mode 100644 index 000000000..db3a33881 --- /dev/null +++ b/applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain.yml @@ -0,0 +1,145 @@ +# DynaCLR-2D-MIP-pretrain (Reef/Kelp smoke test) +# ============================== +# First DynaCLR-2D-MIP training run on the Reef/Kelp (CoreWeave) cluster. +# Same 2D bag-of-channels contrastive recipe as +# applications/dynaclr/configs/training/DynaCLR-2D/bruno/DynaCLR-2D-MIP-BagOfChannels.yml, +# pointed at a mock collection (5 2026_* A549 infectomics experiments, +# Globus-copied to /bio) to validate the Reef training path end to end +# before scaling up to the full production collection. +# +# Launch: +# sbatch applications/dynaclr/configs/training/DynaCLR-2D/reef/DynaCLR-2D-MIP-pretrain-classical.sh +# +# Resume: +# CKPT_PATH=.../last.ckpt sbatch .../DynaCLR-2D-MIP-pretrain-classical.sh + +base: + - ../../recipes/trainer/fit.yml + - ../../recipes/topology/ddp_2gpu.yml + - ../../recipes/model/contrastive_encoder_convnext_tiny.yml + +trainer: + precision: bf16-mixed + max_epochs: 150 + limit_train_batches: 800 + limit_val_batches: 200 + logger: + init_args: + project: DynaCLR-2D-MIP-pretrain + name: null + callbacks: + - class_path: lightning.pytorch.callbacks.LearningRateMonitor + init_args: + logging_interval: step + - class_path: lightning.pytorch.callbacks.ModelCheckpoint + init_args: + monitor: loss/val + every_n_epochs: 1 + save_top_k: 5 + save_last: true + - class_path: viscy_utils.callbacks.OnlineEvalCallback + init_args: + every_n_epochs: 5 + label_key: perturbation + k: 20 + track_id_key: global_track_id + timepoint_key: t + +model: + init_args: + encoder: + init_args: + in_stack_depth: 1 + stem_kernel_size: [1, 4, 4] + stem_stride: [1, 4, 4] + projection_dim: 32 + drop_path_rate: 0.1 + loss_function: + init_args: + temperature: 0.2 + lr: 0.00002 + pca_color_keys: "[perturbation,hours_post_perturbation,experiment,marker]" + log_negative_metrics_every_n_epochs: 2 + example_input_array_shape: [1, 1, 1, 160, 160] + +data: + class_path: dynaclr.data.datamodule.MultiExperimentDataModule + init_args: + cell_index_path: /bio/projects/compimaging/models/collections/DynaCLR-2D-MIP-pretrain-mock.parquet + focus_channel: Phase3D + reference_pixel_size_xy_um: 0.1494 + z_window: 1 + z_extraction_window: 16 + z_focus_offset: 0.3 + yx_patch_size: [256, 256] + final_yx_patch_size: [160, 160] + channels_per_sample: 1 + positive_cell_source: lookup + positive_match_columns: [lineage_id] + positive_channel_source: same + tau_range: [0.5, 2.0] + tau_decay_rate: 2.0 + stratify_by: [perturbation, marker] + split_ratio: 0.8 + batch_size: 256 + num_workers: 4 + prefetch_factor: 1 + buffer_size: 1 + cache_pool_bytes: 0 + file_io_concurrency: 32 + seed: 42 + normalizations: + - class_path: viscy_transforms.NormalizeSampled + init_args: + keys: [channel_0] + level: timepoint_statistics + subtrahend: mean + divisor: std + augmentations: + - class_path: viscy_transforms.BatchedRandAffined + init_args: + keys: [channel_0] + prob: 0.8 + scale_range: [[0.8, 1.3], [0.8, 1.3], [0.8, 1.3]] + rotate_range: [3.14, 0.0, 0.0] + shear_range: [0.05, 0.05, 0.0, 0.05, 0.0, 0.05] + - class_path: viscy_transforms.BatchedRandFlipd + init_args: + keys: [channel_0] + spatial_axes: [1, 2] + prob: 0.5 + - class_path: viscy_transforms.BatchedRandAdjustContrastd + init_args: + keys: [channel_0] + prob: 0.5 + gamma: [0.6, 1.6] + - class_path: viscy_transforms.BatchedRandScaleIntensityd + init_args: + keys: [channel_0] + prob: 0.5 + factors: 0.5 + - class_path: viscy_transforms.BatchedRandGaussianSmoothd + init_args: + keys: [channel_0] + prob: 0.5 + sigma_x: [0.25, 0.50] + sigma_y: [0.25, 0.50] + sigma_z: [0.0, 0.0] + - class_path: viscy_transforms.BatchedRandGaussianNoised + init_args: + keys: [channel_0] + prob: 0.5 + mean: 0.0 + std: 0.1 + # Random Z crop: select 10 of 20 extracted slices for Z-invariance. + # Must come before ZReduction so MIP sees a variable sub-stack. + - class_path: viscy_transforms.BatchedRandSpatialCropd + init_args: + keys: [channel_0] + roi_size: [10, 192, 192] + # Z-reduction: MIP for fluorescence, center-slice for label-free. + # Must be LAST augmentation (before implicit final spatial crop). + - class_path: viscy_transforms.BatchedChannelWiseZReductiond + init_args: + keys: [channel_0] + allow_missing_keys: true diff --git a/applications/dynaclr/configs/training/slurm/train.sh b/applications/dynaclr/configs/training/slurm/train.sh index 5538268cc..6cb8dd42a 100755 --- a/applications/dynaclr/configs/training/slurm/train.sh +++ b/applications/dynaclr/configs/training/slurm/train.sh @@ -38,7 +38,27 @@ function cleanup() { } trap cleanup EXIT -mkdir -p "${RUN_DIR}/checkpoints" +# Scope checkpoints (our own last.ckpt/epoch=*.ckpt, the W&B run id below, AND +# Lightning's own SLURM auto-requeue hpc_ckpt_*.ckpt -- see below) by +# SLURM_JOB_ID so each distinct submission gets its own state instead of +# silently resuming a previous, unrelated run's checkpoint. SLURM preserves +# the same SLURM_JOB_ID across a genuine preemption+requeue (Restarts +# increments, the JobID doesn't), so real resumes still find their own +# checkpoint here; a fresh `sbatch` submission always gets a new SLURM_JOB_ID +# and starts clean. +CKPT_DIR="${RUN_DIR}/checkpoints" +if [ -n "${SLURM_JOB_ID:-}" ]; then + CKPT_DIR="${CKPT_DIR}/${SLURM_JOB_ID}" +fi +mkdir -p "${CKPT_DIR}" + +# #SBATCH --output can't reference $RUN_DIR (SLURM parses #SBATCH directives +# before this script's env vars exist), so redirect stdout/stderr here once +# RUN_DIR is known. Everything from this point on (scontrol, srun, dynaclr +# fit) lands in the run directory instead of the sbatch submission dir. +if [ -n "${SLURM_JOB_ID:-}" ]; then + exec > "${RUN_DIR}/slurm-${SLURM_JOB_ID}.out" 2>&1 +fi # Rotate existing config.yaml before Lightning overwrites it if [ -f "${RUN_DIR}/config.yaml" ]; then @@ -61,19 +81,51 @@ for cfg in $CONFIGS; do CONFIG_FLAGS="${CONFIG_FLAGS} --config ${WORKSPACE_DIR}/${cfg}" done +# Auto-resume after SLURM preemption/requeue: if no explicit CKPT_PATH was +# given but a checkpoint exists in this job's checkpoint dir, resume from it. +# A fresh run (new SLURM_JOB_ID) has no last.ckpt, so it trains from scratch. +if [ -z "${CKPT_PATH:-}" ] && [ -f "${CKPT_DIR}/last.ckpt" ]; then + CKPT_PATH="${CKPT_DIR}/last.ckpt" + echo "Resuming from ${CKPT_PATH}" +fi + CKPT_FLAG="" if [ -n "${CKPT_PATH:-}" ]; then CKPT_FLAG="--ckpt_path ${CKPT_PATH}" fi -WANDB_ID_FLAG="" -if [ -n "${WANDB_RUN_ID:-}" ]; then - WANDB_ID_FLAG="--trainer.logger.init_args.id=${WANDB_RUN_ID} --trainer.logger.init_args.resume=must" +# Persist the W&B run id so a requeued job continues the same run (continuous +# metrics across preemptions). The id is generated once on first launch and +# reused on every resume. Generating it shell-side (rather than reading it back +# from wandb) avoids touching logger.experiment in a callback, which can +# deadlock DDP. +WANDB_ID_FILE="${CKPT_DIR}/.wandb_run_id" +if [ -z "${WANDB_RUN_ID:-}" ]; then + if [ -f "${WANDB_ID_FILE}" ]; then + WANDB_RUN_ID="$(cat "${WANDB_ID_FILE}")" + else + WANDB_RUN_ID="$(uv run --project "$WORKSPACE_DIR" python -c 'import secrets; print(secrets.token_hex(4))')" + echo "${WANDB_RUN_ID}" > "${WANDB_ID_FILE}" + fi fi +# resume=allow (not must) so the first launch can create the run; subsequent +# requeues find the existing id and continue it. +WANDB_ID_FLAG="--trainer.logger.init_args.id=${WANDB_RUN_ID} --trainer.logger.init_args.resume=allow" + +# default_root_dir=CKPT_DIR (not RUN_DIR): Lightning's own SLURMEnvironment +# writes/reads its auto-requeue checkpoint (hpc_ckpt_N.ckpt) directly under +# default_root_dir with no separate config knob -- if this stayed RUN_DIR, a +# brand new SLURM_JOB_ID would silently auto-resume from a DIFFERENT, earlier +# job's leftover hpc_ckpt file. Pointing it at the per-job CKPT_DIR keeps that +# mechanism scoped consistently with our own ModelCheckpoint dirpath (see +# _configure_checkpoint_dirpath in viscy_utils/cli.py, which now just uses +# default_root_dir directly). WandbLogger's save_dir stays RUN_DIR (unscoped) +# since the W&B run itself is meant to persist across preemption+resume. srun uv run --project "$WORKSPACE_DIR" dynaclr fit \ ${CONFIG_FLAGS} \ - --trainer.default_root_dir="${RUN_DIR}" \ + --slurm_auto_requeue \ + --trainer.default_root_dir="${CKPT_DIR}" \ --trainer.logger.init_args.project="${PROJECT}" \ --trainer.logger.init_args.name="${RUN_NAME}" \ --trainer.logger.init_args.save_dir="${RUN_DIR}" \ diff --git a/docs/clusters/reef.md b/docs/clusters/reef.md new file mode 100644 index 000000000..a02c5ae01 --- /dev/null +++ b/docs/clusters/reef.md @@ -0,0 +1,188 @@ +# Reef/Kelp (CoreWeave) — Job Submission Guide + +Reef and Kelp are CZI's CoreWeave-hosted SLURM clusters. They are **beta** and +**preemptible by default** — this is the core thing that makes job scripts +here different from Bruno (our home-institution cluster, which is not +preemptible). Everything below is written for adapting VisCy's existing +Bruno SLURM patterns (`applications/dynaclr/configs/training/slurm/train.sh`, +`applications/dynacell/tools/sbatch_template*.sbatch`, +`applications/cytoland/examples/configs/*/run_*.slurm`) to run on Reef, not +for writing from scratch. + +Source: internal "AI Research Cluster Reef/Kelp User Guide" (work in +progress, expect breaking changes). + +## Bruno vs Reef, at a glance + +| Aspect | Bruno (home institution) | Reef/Kelp (CoreWeave) | +|---|---|---| +| Preemption | Not preemptible | `--qos mid` / `--qos low` **are** preemptible; `--qos dev` and team QOS are not | +| GPU limits | Constraint-based (`--constraint='h200\|h100'`) | `--qos dev` caps you at 8 GPUs; `mid`/`low` uncapped but cost/don't-cost fairshare | +| Partitions | `gpu`, `cpu` | GPU: named reserved pools e.g. `h100_reserved`, `h200_reserved`; CPU: `cpu`, `cpu-turin-gp-l` | +| Filesystem | `/hpc/mydata/`, `/hpc/projects/` | Single filesystem on `/bio`; home is `/mnt/main0/home/` | +| RunAI/CoreWeave PVC data | N/A | Mounted directly at `/mnt/runai-` (same physical storage as RunAI) | +| Cross-cluster data | N/A | Bruno data does **not** auto-sync to Reef — must be copied manually | +| AWS credentials | Manual (`AWS_PROFILE`, etc.) | Auto-provisioned via OIDC on login + compute nodes — **remove** manual AWS env vars from `.bashrc` | +| Job launch | Hand-written `sbatch` scripts | Either hand-written `sbatch`/`srun`, or `slurm_run` (snapshots repo state at submit time) | +| Env manager | `uv` (see root `CLAUDE.md`) | `uv` or `pixi`, same idea | +| Access | Direct SSH | Tailscale-gated; personal login node `login-reef-` | +| Lightning strategy env | `lightning.pytorch.plugins.environments.SLURMEnvironment` | Same — no change needed | + +## Access + +- Cluster access + Tailscale setup is one-time (Okta "Coreweave Slurm + Cluster" request, Tailscale on the `biohub.org` tailnet). Not a per-job + concern. +- Use your **personal** login node (`login-reef-`), not the shared + one — shared login nodes are being deprecated due to reliability issues. +- Interactive GPU node for debugging a script before submitting a batch job: + ```sh + srun -c 8 --mem 96G -N 1 --gres gpu:h100:1 --qos dev -t 1-0 -J interactive --pty zsh -i + ``` +- Interactive CPU node (data prep, quick checks): + ```sh + srun --partition=cpu -c 8 --mem 64G -N 1 --qos dev -t 1-0 -J interactive --pty zsh -i + ``` + +## QOS — read this before choosing one for a training job + +QOS is Reef's stand-in for Bruno's constraint-based scheduling, but it also +controls **preemption**, which Bruno jobs never had to handle: + +- `--qos dev` — max 8 GPUs, non-preemptible. Use for interactive/debug and + short smoke tests, not long training runs (fairshare will eventually + deprioritize you, but you won't be killed mid-run). +- `--qos teamA`-style (team allocation) — non-preemptible, ask if VisCy/CZ + Biohub has one before defaulting to `mid`. +- `--qos mid` — no GPU limit, costs fairshare, **preemptible** by `dev`/team + QOS jobs. +- `--qos low` — no GPU limit, free (no fairshare cost), **preemptible** by + everything above. + +**Implication for VisCy training scripts:** any job submitted at `mid` or +`low` can be killed and requeued at any time. Long DynaCLR/cytoland training +runs on Reef should: +1. Checkpoint frequently (Lightning's `ModelCheckpoint` already does this in + our configs — just make sure the interval is short enough that a + preemption doesn't lose much progress). +2. Resume from checkpoint automatically. `train.sh` already supports this via + `CKPT_PATH` and `WANDB_RUN_ID` — reuse that pattern rather than inventing + a new one. +3. Add `#SBATCH --requeue` so SLURM automatically resubmits the job on + preemption instead of leaving it dead in the queue. Bruno scripts don't + have this flag because it was never needed there. +4. Not assume "job disappeared from `squeue`" means it failed — check + `sacct -j --format=JobID,State,ExitCode` for `PREEMPTED` vs a real + failure, same idea as the completeness-check guidance in the root + `CLAUDE.md`. + +## Partitions + +- GPU: named reserved pools, e.g. `h100_reserved`, `h200_reserved` (confirm + exact names available with `sinfo` on the day — Reef is still adding + pools). This replaces Bruno's `--partition=gpu --constraint='h200|h100'` + pattern; on Reef pick the partition itself instead of constraining within + the `gpu` partition. +- CPU-only: `cpu` (has a default QOS) or `cpu-turin-gp-l` (no default QOS — + **you must pass `--qos` explicitly** or the job is rejected). Use for + dataset prep / ETL / lightweight inference, same role as CPU-only Bruno + jobs. `cpu-turin-gp-l` runs with `OverSubscribe=NO`, i.e. CPUs are + exclusively allocated (no oversubscription like some Bruno CPU nodes). + +## Filesystem & data paths + +- Single filesystem on `/bio`; user home is `/mnt/main0/home/` (this is + where `slurm_run` also drops job scripts and logs by default: + `/mnt/main0/home//slurm//slurm-.out`). +- CoreWeave/RunAI PVC-backed datasets are mounted at + `/mnt/runai-`, e.g. a `dynamic-imaging-models-120t` PVC is at + `/mnt/runai-dynamic-imaging-models-120t`. This is the same physical storage + RunAI jobs used — no copy needed if the data already lives there. +- **Bruno data is not auto-synced to Reef.** Any dataset referenced by + `WORKSPACE_DIR`/`MODEL_ROOT`-style paths in our Bruno scripts + (`/hpc/mydata/...`, `/hpc/projects/...`) must be manually copied to `/bio` + or the appropriate `/mnt/runai-*` PVC before a Reef job can read it. Don't + assume a path that works on Bruno resolves on Reef. + +## Environment setup + +- `uv` (already our standard, see root `CLAUDE.md`) or `pixi` both work. + Symlink the cache out of `$HOME` the same way we do on Bruno-style HPC: + ```sh + mkdir -p /bio///.cache/uv && ln -s /bio///.cache/uv ~/.cache/uv + ``` + (adjust the target to wherever project storage lands on `/bio` or the + relevant `/mnt/runai-*` PVC — installing envs on a login node is slow, do + it from a compute node if it's dragging.) +- AWS credentials are auto-provisioned via OIDC on both login and compute + nodes. **Remove** any `AWS_PROFILE`/`aws-oidc` setup from `.bashrc` — + leftover Bruno-style AWS env vars can shadow the auto-provisioned ones and + break S3 access. +- **Nextflow**: unlike Bruno (`module load nextflow/24.10.5`), Reef has no + `nextflow` module — install it yourself via `micromamba` (already present + on Reef login nodes; no root needed): + ```sh + micromamba create -y -n nextflow -c bioconda -c conda-forge "nextflow=24.10.5" + ``` + Pin the version to `24.10.5` to match Bruno's module and this repo's + `applications/dynaclr/nextflow/` DAGs — the latest bioconda build (26.x) + turns on Nextflow's "strict parser" by default, which rejects the + `-entry ` flag these DAGs rely on + (`ERROR ~ The '-entry' option is not supported with the strict parser`). + Run it via `micromamba run -n nextflow nextflow run ...` or + `micromamba activate nextflow` first. + +## Two ways to launch: `slurm_run` vs hand-written `sbatch` + +**`slurm_run`** (from `github.com/evolutionaryscale/slurm_run`, install via +`git clone` + `make install` — not yet a pip package) snapshots the repo +state at submit time, so it's a good fit for one-off training/eval launches +where you want the exact commit reproducible. For a VisCy job: +```sh +slurm_run submit \ + --venv=uv \ + --cpus= \ + --gpus= # 1-gpu-per-task; sets the number of tasks \ + --partition= \ + --qos= \ + -- uv run dynaclr fit --config +``` +`slurm_run jobs` lists your recent submissions. + +**Hand-written `sbatch`/`srun` scripts** are still the right choice when you +need our existing patterns — `train.sh`'s config-copying/checkpoint-resume +logic, or `dynacell`'s NCCL preflight smoke test — that `slurm_run` doesn't +know about. When adapting one of our Bruno `.slurm` files for Reef, change: +- `#SBATCH --partition=...` to a Reef GPU/CPU partition name (see above). +- `#SBATCH --constraint=...` (Bruno-only) → drop it; pick the partition + instead. +- Add `#SBATCH --qos=` (Reef requires an explicit QOS; Bruno + scripts often didn't set one). +- Add `#SBATCH --requeue` if running at `mid`/`low`, so preemption resubmits + instead of dying silently. +- Update any `/hpc/mydata/...` or `/hpc/projects/...` path to its `/bio` or + `/mnt/runai-*` equivalent (see Filesystem section). +- Keep `--ntasks-per-node=N` (not `--ntasks=N`) for any Lightning DDP job — + this is a Lightning `SLURMEnvironment` requirement, not Bruno- or + Reef-specific, and the existing invariant in the root `CLAUDE.md` (must + match `trainer.devices` and `--gpus`/`--gpus-per-node`) still applies + unchanged on Reef. + +## Monitoring + +- `slurm_run jobs`, or standard `squeue`/`sacct`, work as usual. +- Web dashboards: "All Jobs Metrics" (click a Job ID for per-job metrics) and + "Cluster Utilization" — both beta, report bad numbers to Sashidhar Guntury. +- Reminder from root `CLAUDE.md` still applies here, and matters *more* on + Reef: `wandb` `state: finished` doesn't distinguish a clean finish from a + preempted/`scancel`'d run. On Reef, also check `sacct` for `PREEMPTED` + specifically before assuming a dead job failed outright. + +## Current rollout limits (beta caveats) + +As of this writing Reef only has confirmed support for: single-GPU jobs, +single-node multi-GPU (DDP), small multi-node (2 nodes × 8 GPUs), +checkpoint+resume, and W&B logging — which covers current VisCy training +jobs. Large-scale sweeps, queue-based worker patterns, and general ETL +pipelines on Reef are still on the roadmap (not yet documented) — don't +assume they work without checking `sinfo`/the guide for updates first. diff --git a/packages/viscy-utils/src/viscy_utils/cli.py b/packages/viscy-utils/src/viscy_utils/cli.py index 8d9e6b2ad..fd28f1b8b 100644 --- a/packages/viscy-utils/src/viscy_utils/cli.py +++ b/packages/viscy-utils/src/viscy_utils/cli.py @@ -4,6 +4,7 @@ import logging import os import re +import signal import sys import tempfile from collections.abc import Callable @@ -12,15 +13,17 @@ import torch import yaml -from jsonargparse import Namespace, lazy_instance +from jsonargparse import Namespace, lazy_instance, namespace_to_dict from lightning.pytorch import LightningDataModule, LightningModule -from lightning.pytorch.cli import LightningCLI -from lightning.pytorch.loggers import WandbLogger +from lightning.pytorch.cli import LightningCLI, SaveConfigCallback +from lightning.pytorch.loggers import Logger, WandbLogger +from lightning.pytorch.plugins.environments import SLURMEnvironment from viscy_utils.compose import load_composed_config from viscy_utils.trainer import VisCyTrainer _WANDB_LOGGER_CLASS_PATH = "lightning.pytorch.loggers.WandbLogger" +_MODEL_CHECKPOINT_CLASS_PATH = "lightning.pytorch.callbacks.ModelCheckpoint" _WANDB_RUN_NAME_PREFIX = re.compile(r"^\d{8}-\d{6}_") _WANDB_RUN_TIMESTAMP_FORMAT = r"%Y%m%d-%H%M%S" @@ -69,6 +72,115 @@ def _configure_wandb_logger( init_args["group"] = base_name +def _configure_slurm_requeue(config: Namespace, subcommand: str | None) -> None: + """Attach :class:`SLURMEnvironment` with auto-requeue under SLURM batch jobs. + + On a preemptible cluster, SLURM sends ``SIGUSR1`` before killing the job. + With :class:`SLURMEnvironment` attached, Lightning catches that signal, + writes a checkpoint, and calls ``scontrol requeue`` so the job resumes + from the checkpoint when resources free up. + + Opt in with ``--slurm_auto_requeue``; when the flag is absent, normal + Lightning behavior is kept. Even when set, the plugin is attached only if + ``SLURMEnvironment.detect()`` is true, so passing it off SLURM (or in an + interactive allocation without it) is a no-op. + """ + root = config[subcommand] if subcommand is not None else config + if not isinstance(root, Namespace): + return + if not root.get("slurm_auto_requeue", False): + return + if not SLURMEnvironment.detect(): + return + trainer = root.get("trainer") + if not isinstance(trainer, Namespace): + return + plugins = trainer.get("plugins") + if plugins is None: + plugins = [] + elif not isinstance(plugins, list): + plugins = [plugins] + if any(isinstance(p, SLURMEnvironment) for p in plugins): + return + # NOTE: append an already-instantiated object, not a class_path/init_args + # spec. before_instantiate_classes() runs after jsonargparse has already + # resolved Union-typed slots like trainer.plugins from raw specs into + # concrete values, so a raw spec inserted here fails validation ("Expected + # a . Got value: [Namespace(...)]"). This does mean + # jsonargparse can't serialize the object back out when SaveConfigCallback + # dumps the config, and warns "Unable to serialize instance ..." -- cosmetic, + # but a real (harmless) side effect of this hook running where it does. + plugins.append(SLURMEnvironment(auto_requeue=True, requeue_signal=signal.SIGUSR1)) + trainer["plugins"] = plugins + + +def _configure_checkpoint_dirpath(config: Namespace, subcommand: str | None) -> None: + """Pin ``ModelCheckpoint.dirpath`` to ``default_root_dir``. + + Without this, Lightning's default resolution (``ModelCheckpoint. + __resolve_ckpt_dir``) nests checkpoints under + ``{logger.save_dir}/{logger.name}/{logger.version}/checkpoints`` whenever + a logger is attached -- for ``WandbLogger`` that's the run's random hex + id, so checkpoints land in a differently-named subfolder every run while + ``default_root_dir`` itself stays empty. + + This intentionally does NOT add its own "checkpoints" subfolder or any + SLURM_JOB_ID scoping -- `train.sh` already passes a ``default_root_dir`` + that's exactly the intended checkpoint directory (see its ``CKPT_DIR``, + scoped by SLURM_JOB_ID so a fresh submission doesn't silently resume a + previous job's state, and matching where Lightning's own SLURMEnvironment + reads/writes its ``hpc_ckpt_*.ckpt`` -- that path is hardcoded to + ``default_root_dir`` with no separate config knob, so this hook and + `train.sh` must agree on what ``default_root_dir`` means). Any caller + that wants a nested "checkpoints" subfolder should pass that directly as + ``default_root_dir``. + + A CLI override like ``--trainer.callbacks.1.init_args.dirpath=...`` isn't + supported by jsonargparse for typed ``list[Callback]`` unions, so this is + done as a config rewrite instead -- matching :func:`_configure_wandb_logger` + and :func:`_configure_slurm_requeue`. + """ + root = config[subcommand] if subcommand is not None else config + if not isinstance(root, Namespace): + return + trainer = root.get("trainer") + if not isinstance(trainer, Namespace): + return + default_root_dir = trainer.get("default_root_dir") + if not default_root_dir: + return + callbacks = trainer.get("callbacks") + if callbacks is None: + return + callback_list = callbacks if isinstance(callbacks, list) else [callbacks] + for callback in callback_list: + if not isinstance(callback, Namespace): + continue + if callback.get("class_path") != _MODEL_CHECKPOINT_CLASS_PATH: + continue + init_args = callback.get("init_args") + if not isinstance(init_args, Namespace): + init_args = Namespace() + callback["init_args"] = init_args + if init_args.get("dirpath") is None: + init_args["dirpath"] = default_root_dir + + +class VisCySaveConfigCallback(SaveConfigCallback): + """Also push the full resolved config (trainer/model/data) to the logger. + + The default ``SaveConfigCallback`` only writes ``config.yaml`` to the run + directory; ``save_config`` (its designated extension point, see the base + class's own docstring) is a no-op, so none of it reaches the logger's own + config UI -- e.g. wandb's run config panel only ever sees wandb's own + telemetry, not our trainer/model/data hyperparameters. + """ + + def save_config(self, trainer, pl_module, stage: str) -> None: # noqa: D102 + if isinstance(trainer.logger, Logger): + trainer.logger.log_hyperparams(namespace_to_dict(self.config)) + + class VisCyCLI(LightningCLI): """Extending lightning CLI arguments and defaults.""" @@ -84,12 +196,22 @@ def subcommands() -> dict[str, set[str]]: return subcommands def add_arguments_to_parser(self, parser) -> None: - """Set default logger.""" + """Set default logger and SLURM auto-requeue toggle.""" parser.set_defaults( { "trainer.logger": lazy_instance(WandbLogger), } ) + parser.add_argument( + "--slurm_auto_requeue", + action="store_true", + help=( + "Opt in to SLURMEnvironment(auto_requeue=True): when running " + "under SLURM, preempted jobs checkpoint and requeue " + "automatically. Absent (default) keeps normal Lightning " + "behavior. No effect off SLURM." + ), + ) def _parse_ckpt_path(self) -> None: # For predict/test/validate: snapshot model init_args before checkpoint @@ -127,6 +249,8 @@ def _parse_ckpt_path(self) -> None: def before_instantiate_classes(self) -> None: """Apply shared config rewrites before Lightning object creation.""" _configure_wandb_logger(self.config, self.subcommand) + _configure_slurm_requeue(self.config, self.subcommand) + _configure_checkpoint_dirpath(self.config, self.subcommand) def _setup_environment() -> None: @@ -224,6 +348,7 @@ def main(*, resolver: Callable[[dict], dict] | None = None) -> None: seed_everything_default=42, subclass_mode_model=require_model, subclass_mode_data=require_data, + save_config_callback=VisCySaveConfigCallback, save_config_kwargs={"overwrite": True}, parser_kwargs={"description": "Computer vision models for single-cell phenotyping."}, ) diff --git a/uv.lock b/uv.lock index 9b52e7d2d..84ed939d9 100644 --- a/uv.lock +++ b/uv.lock @@ -1394,8 +1394,10 @@ dependencies = [ eval = [ { name = "anndata" }, { name = "dtaidistance" }, + { name = "joblib" }, { name = "natsort" }, { name = "phate" }, + { name = "pot" }, { name = "scikit-learn" }, { name = "statsmodels" }, { name = "umap-learn" }, @@ -1440,10 +1442,12 @@ requires-dist = [ { name = "gurobipy", marker = "extra == 'tracking'", specifier = ">=12.0.1,<13" }, { name = "imageio-ffmpeg", marker = "extra == 'viz'" }, { name = "iohub", specifier = ">=0.3.6" }, + { name = "joblib", marker = "extra == 'eval'" }, { name = "matplotlib", marker = "extra == 'viz'" }, { name = "natsort", marker = "extra == 'eval'" }, { name = "onnxruntime-gpu", marker = "extra == 'tracking'" }, { name = "phate", marker = "extra == 'eval'" }, + { name = "pot", marker = "extra == 'eval'" }, { name = "py-ctcmetrics", marker = "extra == 'tracking'" }, { name = "pytorch-metric-learning" }, { name = "pyyaml" }, @@ -4479,6 +4483,32 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/2a/2d/d4bf65e47cea8ff2c794a600c4fd1273a7902f268757c531e0ee9f18aa58/pooch-1.9.0-py3-none-any.whl", hash = "sha256:f265597baa9f760d25ceb29d0beb8186c243d6607b0f60b83ecf14078dbc703b", size = 67175, upload-time = "2026-01-30T19:15:08.36Z" }, ] +[[package]] +name = "pot" +version = "0.9.6.post1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "numpy" }, + { name = "scipy" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/42/8b/5f939eaf1fbeb7ff914fe540d659486951a056e5537b8f454362045b6c72/pot-0.9.6.post1.tar.gz", hash = "sha256:9b6cc14a8daecfe1268268168cf46548f9130976b22b24a9e8ec62a734be6c43", size = 604243, upload-time = "2025-09-22T12:51:14.894Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b9/28/13622807461f9f6082a8cd6768f9b4a810bc3a8fda474b81572da94b4d23/pot-0.9.6.post1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:f7c542fc20662e35c24dd82eeff8a737220757434d7f0038664a7322221452f7", size = 599240, upload-time = "2025-09-22T12:50:44.848Z" }, + { url = "https://files.pythonhosted.org/packages/c6/5c/b4e017560531f53d06798c681b0d0a9488bb8116bc98da9d399a3d096391/pot-0.9.6.post1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:c1755516a7354cbd6110ad2e5f341b98b9968240c2f0f67b0ff5e3ebcb3105bd", size = 464695, upload-time = "2025-09-22T12:50:46.341Z" }, + { url = "https://files.pythonhosted.org/packages/07/9f/57e49b3f7173359741053c5e2766a45dcf649d767c2e967ef93526c9045f/pot-0.9.6.post1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:f3207362d3e3b5aaa783f452aa85f66e83edbefb5764f34662860af54ac72ee6", size = 454726, upload-time = "2025-09-22T12:50:47.953Z" }, + { url = "https://files.pythonhosted.org/packages/30/60/fa72dd6094f7dbe6b38e2c6907af8cd0f18c6bd107e0cf4874deddaba883/pot-0.9.6.post1-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:05f6659c5657e6d7e9f98f4a82e0ed64f88e9fce69b2e557416d156343919ba3", size = 1503391, upload-time = "2025-09-22T12:50:49.336Z" }, + { url = "https://files.pythonhosted.org/packages/2f/3f/cc519c1176116271b6282268a705162fa042c16cc922bc56039445c9d697/pot-0.9.6.post1-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:4f1b0148ae17bec0ed12264c6da3a05e13913b716e2a8c9043242b5d8349d8df", size = 1528170, upload-time = "2025-09-22T12:50:50.625Z" }, + { url = "https://files.pythonhosted.org/packages/f5/01/0132c94404cd0b1b2f21c4a49698db9dcd6107c47c02b22df1ed38206b2a/pot-0.9.6.post1-cp312-cp312-win32.whl", hash = "sha256:571e543cc2b0a462365002203595baf2b89c3d064cce4fce70fd1231e832c21f", size = 440577, upload-time = "2025-09-22T12:50:51.716Z" }, + { url = "https://files.pythonhosted.org/packages/c1/6d/23229c0e198a4f7fb27750b3ef8497e6ebed23fe531ed64b5194da8b2b02/pot-0.9.6.post1-cp312-cp312-win_amd64.whl", hash = "sha256:b1d8bd9a334c72baa37f9a2b268de5366c23c0f9c9e3d6dc25d150137ec2823c", size = 455404, upload-time = "2025-09-22T12:50:52.956Z" }, + { url = "https://files.pythonhosted.org/packages/53/17/e4aebb8deef58b0d40ac339d952d12c63559801b50ae43c622d49bebda7e/pot-0.9.6.post1-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:659fff750a162f58b52b33a64c4ac358f4ff44e9dff0841052c088e1b6a54430", size = 596485, upload-time = "2025-09-22T12:50:54.309Z" }, + { url = "https://files.pythonhosted.org/packages/f7/b9/3646c153b13f999ac30112dcf85c5f233af79b0d98c37b52dda9a624c91b/pot-0.9.6.post1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:4f54830e9f9cb78b1ff7abd5c5bf162625ed6aea903241267c64ea9f0fb73ddb", size = 463244, upload-time = "2025-09-22T12:50:56.004Z" }, + { url = "https://files.pythonhosted.org/packages/53/e9/c7092f7aec8cb32739ad66ba1f1259626546e4893b61b905ce2da3987235/pot-0.9.6.post1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:e9fd4b1fafacd37debdb984687ddb26f5c43d1429401847d388a6f1bd1f10e98", size = 453215, upload-time = "2025-09-22T12:50:57.515Z" }, + { url = "https://files.pythonhosted.org/packages/0c/a1/f0187ab15aa1538ece07b28f3a7938b8592ef01fbe37b1a8f9c2f8f47f4d/pot-0.9.6.post1-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:ec097ec0ef8bb93fee8cdb187b6a0a9653613cba7b06bb603247930e2c629cdc", size = 1496245, upload-time = "2025-09-22T12:50:58.848Z" }, + { url = "https://files.pythonhosted.org/packages/29/fa/85af71553b7e990fc37da8d5f2e7294ec66297e62cba419efeec11518e5a/pot-0.9.6.post1-cp313-cp313-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:299f11f172908d799793ef18b2bc82452305350d2528d243e255a17876e98a57", size = 1521691, upload-time = "2025-09-22T12:51:00.203Z" }, + { url = "https://files.pythonhosted.org/packages/19/ae/96b2bce173b3d2d3d0faf8b7362fe79e60e1a6a939c9459b2f7b64e625d8/pot-0.9.6.post1-cp313-cp313-win32.whl", hash = "sha256:8a1d95310faae9c75355d9e2fac8dfac41316a2450061eefc982ee498a687a34", size = 439760, upload-time = "2025-09-22T12:51:01.601Z" }, + { url = "https://files.pythonhosted.org/packages/f7/b1/8ca34418e7c4a2ec666e2204539577287223c4e78ab80b1c746cedb559c3/pot-0.9.6.post1-cp313-cp313-win_amd64.whl", hash = "sha256:a43e2b61389bd32f5b488da2488999ed55867e95fedb25dd64f9f390e40b4fab", size = 454228, upload-time = "2025-09-22T12:51:03.215Z" }, +] + [[package]] name = "prometheus-client" version = "0.25.0" @@ -6892,6 +6922,7 @@ all = [ { name = "tensorstore" }, { name = "tifffile" }, { name = "torchvision" }, + { name = "viscy-transforms" }, ] livecell = [ { name = "pycocotools" }, @@ -6903,6 +6934,7 @@ mmap = [ ] triplet = [ { name = "tensorstore" }, + { name = "viscy-transforms" }, ] [package.dev-dependencies] @@ -6941,6 +6973,8 @@ requires-dist = [ { name = "torchvision", marker = "extra == 'all'" }, { name = "torchvision", marker = "extra == 'livecell'" }, { name = "tqdm" }, + { name = "viscy-transforms", marker = "extra == 'all'", editable = "packages/viscy-transforms" }, + { name = "viscy-transforms", marker = "extra == 'triplet'", editable = "packages/viscy-transforms" }, { name = "zarr" }, ] provides-extras = ["all", "livecell", "mmap", "triplet"]