From ceea2c30276be326b4c26c625f834b6e8fcb04b6 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Mon, 1 Jun 2026 16:09:44 -0400 Subject: [PATCH 01/46] feat(backend): add a Ray Jobs API execution backend Add an ExecutionBackend abstraction plus a RayBackend that submits pipeline stages to a pre-provisioned Ray cluster over the HTTP Ray Jobs API, selected via cluster_config (backend.name: ray + dashboard_url). - backends.py (new): ExecutionBackend interface + get_execution_backend() resolver (default/none/ray + kubernetes-ray aliases + legacy with_ray compat), selector/metadata helpers. - ray_backend.py (new): RayBackend - dashboard-url normalization, preflight checks, concurrent dependency-ordered Jobs-API submission, manifest.jsonl + per-job logs. - exp.py: backend env-overrides + ray-precreated gate in get_executor; add_task stage_metadata + queue_ray_job_commands; populate command_images at each command site (fixes dead image-label selection); run_exp backend delegation. - declarative.py: resolve the backend and queue Ray Jobs on the RL Pipeline path (used by the rollout/verify GRPO stages). - __init__.py: export backend symbols. - tests/test_backends.py (new): selector/metadata unit tests. - cluster_configs/example-slurm-ray-k8s-precreated.yaml (new): reference. Signed-off-by: Nick Gupta --- .../example-slurm-ray-k8s-precreated.yaml | 121 +++ nemo_skills/pipeline/utils/__init__.py | 7 + nemo_skills/pipeline/utils/backends.py | 303 +++++++ nemo_skills/pipeline/utils/declarative.py | 34 +- nemo_skills/pipeline/utils/exp.py | 134 ++- nemo_skills/pipeline/utils/ray_backend.py | 824 ++++++++++++++++++ tests/test_backends.py | 146 ++++ 7 files changed, 1549 insertions(+), 20 deletions(-) create mode 100644 cluster_configs/example-slurm-ray-k8s-precreated.yaml create mode 100644 nemo_skills/pipeline/utils/backends.py create mode 100644 nemo_skills/pipeline/utils/ray_backend.py create mode 100644 tests/test_backends.py diff --git a/cluster_configs/example-slurm-ray-k8s-precreated.yaml b/cluster_configs/example-slurm-ray-k8s-precreated.yaml new file mode 100644 index 0000000000..6a3e4cfae5 --- /dev/null +++ b/cluster_configs/example-slurm-ray-k8s-precreated.yaml @@ -0,0 +1,121 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# This config is a sample for running NeMo Skills and NVFlow recipes against a +# pre-created Ray cluster running on Kubernetes. +# +# Notes: +# - SLURM remains the scheduler used by nemo_run. +# - The execution backend is Ray; jobs connect to an existing Ray endpoint. +# - Replace all CHANGE_ME values before use. + +executor: slurm + +containers: + # Point to images available on your SLURM cluster. + trtllm: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc8 + vllm: nvcr.io/nvidia/pytorch:25.02-py3 + sglang: lmsysorg/sglang:v0.5.10.post1 + megatron: nvcr.io/nvidia/pytorch:25.02-py3 + sandbox: nvcr.io/nvidia/pytorch:25.02-py3 + nemo-skills: nvcr.io/nvidia/pytorch:25.02-py3 + verl: nvcr.io/nvidia/pytorch:25.02-py3 + nemo-rl: nvcr.io/nvidia/pytorch:25.02-py3 + +job_name_prefix: "nemo_skills:" + +# If launching from outside cluster, configure ssh_tunnel and remove top-level job_dir. +# ssh_tunnel: +# host: CHANGE_ME.cluster.company.com +# user: CHANGE_ME +# job_dir: /lustre/CHANGE_ME/nemo_run/jobs +# identity: /home/CHANGE_ME/.ssh/id_rsa + +# If launching from within cluster, use job_dir directly. +job_dir: /lustre/CHANGE_ME/nemo_run/jobs + +account: CHANGE_ME_ACCOUNT +partition: CHANGE_ME_GPU_PARTITION +cpu_partition: CHANGE_ME_CPU_PARTITION + +# Optional timeout mapping by partition. +default_timeout: "2-00:00:00" +timeouts: + CHANGE_ME_GPU_PARTITION: "2-00:00:00" + CHANGE_ME_CPU_PARTITION: "1-00:00:00" + +# Backend config: pre-created Ray cluster on Kubernetes. +backend: + name: ray + control_plane: kubernetes + precreated_cluster: true + # Ray Client endpoint to connect to existing cluster. + endpoint: ray://ray-head.ray.svc.cluster.local:10001 + kubernetes: + # offline: batch style, online: service style lifecycle hints + mode: offline + # Optional labels forwarded to Ray job entrypoint placement for pre-created clusters. + # Example equivalent in Ray client API: + # client.submit_job(..., entrypoint_label_selector={"type": "worker"}) + # entrypoint_label_selector: + # type: worker + # Optionally append a label derived from the selected NeMo Skills container image. + # image_label_key: nemo/image + # Optionally provide explicit labels keyed by container name from cluster_config.containers. + # These labels are merged into entrypoint_label_selector for ray job submit. + # image_label_selectors: + # nemo-skills: + # nemo/workload: skills + # nemo-rl: + # key: nemo/workload + # value: rl + + # Optional preflight checks run before Ray-backed experiment start. + # These are useful for failing fast when endpoint, labels, or cluster capacity + # are not aligned with the job requirements. + # preflight: + # enabled: true + # # If true, require Ray endpoint connectivity and at least one live node. + # require_ray_endpoint: true + # # If true, fail if label inspection cannot be performed. + # strict_label_check: true + # # Require at least one live node matching each key/value pair. + # required_node_labels: + # - key: nemo/has-nemo-rl + # value: "true" + # - key: workload + # value: "training" + # # Require minimum total live-cluster resources before submission. + # # Supported aliases: gpu|gpus -> GPU, cpu|cpus -> CPU, mem|memory -> memory. + # min_cluster_resources: + # gpu: 16 + # cpu: 64 + +# Required mounts for models/data/workspace. +mounts: + - /lustre/CHANGE_ME/models:/models + - /lustre/CHANGE_ME/data:/data + - /lustre/CHANGE_ME/workspace:/workspace + +# HF_HOME must be mounted when skip_hf_home_check is false (default). +env_vars: + - HF_HOME=/models/hf-cache + # Optional but often useful for large checkpoint IO. + - NCCL_DEBUG=warn + - TORCH_DISTRIBUTED_DEBUG=off + +# Optional extra vars to pass through from launcher environment: +# required_env_vars: +# - WANDB_API_KEY +# - NVIDIA_API_KEY diff --git a/nemo_skills/pipeline/utils/__init__.py b/nemo_skills/pipeline/utils/__init__.py index 7d46b65f7f..dc1fd287ae 100644 --- a/nemo_skills/pipeline/utils/__init__.py +++ b/nemo_skills/pipeline/utils/__init__.py @@ -14,6 +14,13 @@ # importing every utility function here to make them available in the pipeline.utils namespace +from nemo_skills.pipeline.utils.backends import ( + BackendRunOptions, + ExecutionBackend, + get_execution_backend, + stop_stage_tasks, + track_stage_tasks, +) from nemo_skills.pipeline.utils.cluster import ( _get_tunnel_cached, cluster_download_dir, diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py new file mode 100644 index 0000000000..1fb0a829ae --- /dev/null +++ b/nemo_skills/pipeline/utils/backends.py @@ -0,0 +1,303 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import Any, Dict + +import nemo_run as run + +from nemo_skills.utils import get_logger_name + +LOG = logging.getLogger(get_logger_name(__file__)) + + +def _normalize_backend_config(cluster_config: Dict[str, Any]) -> Dict[str, Any]: + """Normalize backend config from multiple compatible keys.""" + backend_config = cluster_config.get("backend") or cluster_config.get("execution_backend") or {} + if isinstance(backend_config, str): + return {"name": backend_config} + if not isinstance(backend_config, dict): + raise ValueError( + f"cluster_config backend must be a dict or string, got {type(backend_config).__name__}" + ) + return backend_config + + +def _resolve_selector_keys_with_container_map( + selectors: Dict[str, Any] | None, + containers: Dict[str, Any] | None, +) -> Dict[str, Any] | None: + """Resolve image selector keys from container names when possible. + + If a selector key matches a key in cluster_config["containers"], it is replaced + by that container's configured image/path. Non-matching keys are preserved as-is + to retain compatibility with explicit image/path or glob keys. + """ + if selectors is None or not isinstance(selectors, dict): + return selectors + + if not isinstance(containers, dict): + containers = {} + + resolved: Dict[str, Any] = {} + for key, value in selectors.items(): + key_str = str(key) + resolved_key = str(containers.get(key_str, key_str)) + resolved[resolved_key] = value + return resolved + + +@dataclass +class BackendRunOptions: + sequential: bool = False + dry_run: bool = False + + +class ExecutionBackend: + """Execution backend hook interface. + + Backends may customize script metadata and lifecycle operations for experiments. + """ + + name = "default" + + def stage_metadata( + self, + *, + use_with_ray_cluster: bool = False, + container_image: str | None = None, + ) -> Dict[str, Any] | None: + if use_with_ray_cluster: + return {"use_with_ray_cluster": True} + return None + + def get_env_overrides(self) -> Dict[str, str]: + return {} + + def start_experiment(self, exp: run.Experiment, cluster_config: Dict[str, Any], options: BackendRunOptions): + if options.dry_run: + LOG.info("Dry run mode is enabled, not running the experiment.") + return + + if cluster_config["executor"] != "slurm": + exp.run(detach=False, tail_logs=True, sequential=options.sequential) + else: + exp.run(detach=True, sequential=options.sequential) + + def track_experiment(self, exp: run.Experiment, include_finished: bool = True) -> Dict[str, Any]: + status_dict = exp.status(return_dict=True) + if include_finished: + return status_dict + + active_states = { + "RUNNING", + "PENDING", + "SUBMITTED", + "UNKNOWN", + } + return { + task_name: info + for task_name, info in status_dict.items() + if str(info.get("status", "")).split(".")[-1] in active_states + } + + def stop_experiment(self, exp: run.Experiment, only_active: bool = True) -> list[str]: + cancelled_jobs = [] + active_states = { + "RUNNING", + "PENDING", + "SUBMITTED", + "UNKNOWN", + } + status_map = exp.status(return_dict=True) + for task_name, info in status_map.items(): + state = str(info.get("status", "")).split(".")[-1] + handle = info.get("handle") + if not handle: + continue + if only_active and state not in active_states: + continue + exp.cancel(handle) + cancelled_jobs.append(task_name) + + return cancelled_jobs + + +# --------------------------------------------------------------------------- +# RayBackend lives in ray_backend.py – imported here for backwards compat. +# --------------------------------------------------------------------------- +from nemo_skills.pipeline.utils.ray_backend import RayBackend # noqa: E402 # re-exported + + +class _RayBackendShim(RayBackend): + """Shim so that isinstance checks against the old import path still work.""" + + +# Keep the name RayBackend pointing at the canonical class. +del _RayBackendShim # only the alias is needed + + +# Placeholder so linters don't complain about the import being "unused". +__all__ = [ + "BackendRunOptions", + "ExecutionBackend", + "RayBackend", + "get_execution_backend", + "track_stage_tasks", + "stop_stage_tasks", +] + +def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = False) -> ExecutionBackend: + """Resolve execution backend from cluster config and compatibility flags. + + Resolution priority: + 1. Explicit backend/ execution_backend in cluster config + 2. Legacy with_ray compatibility flag + 3. Default backend + """ + backend_config = _normalize_backend_config(cluster_config) + backend_name = str(backend_config.get("name") or "").strip().lower() + legacy_ray_endpoint = cluster_config.get("ray_endpoint") + + if not backend_name: + if with_ray: + return RayBackend(endpoint=legacy_ray_endpoint, precreated_cluster=bool(legacy_ray_endpoint)) + return ExecutionBackend() + + if backend_name in {"default", "none"}: + return ExecutionBackend() + if backend_name == "ray": + endpoint = ( + backend_config.get("endpoint") + or backend_config.get("ray_endpoint") + or (backend_config.get("kubernetes") or {}).get("endpoint") + or legacy_ray_endpoint + ) + dashboard_url = ( + backend_config.get("dashboard_url") + or backend_config.get("jobs_api_url") + or (backend_config.get("kubernetes") or {}).get("dashboard_url") + ) + precreated_cluster = bool(backend_config.get("precreated_cluster", False)) or bool(endpoint) + control_plane = str(backend_config.get("control_plane") or "").strip().lower() + k8s_cfg = backend_config.get("kubernetes") or {} + selector = backend_config.get("entrypoint_label_selector") or k8s_cfg.get("entrypoint_label_selector") + image_label_key = backend_config.get("image_label_key") or k8s_cfg.get("image_label_key") + image_label_selectors = backend_config.get("image_label_selectors") or k8s_cfg.get("image_label_selectors") + image_label_selectors = _resolve_selector_keys_with_container_map( + image_label_selectors, cluster_config.get("containers") + ) + if control_plane == "kubernetes": + mode = k8s_cfg.get("mode", "offline") + endpoint = endpoint or k8s_cfg.get("endpoint") + return RayBackend( + endpoint=endpoint, + precreated_cluster=bool(endpoint), + control_plane="kubernetes", + kubernetes_mode=mode, + dashboard_url=dashboard_url, + entrypoint_label_selector=selector, + image_label_key=image_label_key, + image_label_selectors=image_label_selectors, + ) + return RayBackend( + endpoint=endpoint, + precreated_cluster=precreated_cluster, + control_plane=control_plane or None, + dashboard_url=dashboard_url, + entrypoint_label_selector=selector, + image_label_key=image_label_key, + image_label_selectors=image_label_selectors, + ) + if backend_name in {"kubernetes-ray", "ray-kubernetes", "ray_kubernetes"}: + k8s_cfg = backend_config.get("kubernetes") or {} + mode = k8s_cfg.get("mode", "offline") + selector = backend_config.get("entrypoint_label_selector") or k8s_cfg.get("entrypoint_label_selector") + image_label_key = backend_config.get("image_label_key") or k8s_cfg.get("image_label_key") + image_label_selectors = backend_config.get("image_label_selectors") or k8s_cfg.get("image_label_selectors") + image_label_selectors = _resolve_selector_keys_with_container_map( + image_label_selectors, cluster_config.get("containers") + ) + endpoint = ( + backend_config.get("endpoint") + or k8s_cfg.get("endpoint") + or backend_config.get("ray_endpoint") + or legacy_ray_endpoint + ) + dashboard_url = ( + backend_config.get("dashboard_url") + or backend_config.get("jobs_api_url") + or k8s_cfg.get("dashboard_url") + ) + return RayBackend( + endpoint=endpoint, + precreated_cluster=bool(endpoint), + control_plane="kubernetes", + kubernetes_mode=mode, + dashboard_url=dashboard_url, + entrypoint_label_selector=selector, + image_label_key=image_label_key, + image_label_selectors=image_label_selectors, + ) + + raise ValueError( + f"Unsupported execution backend '{backend_name}'. Supported backends: default, ray, kubernetes-ray" + ) + + +def _with_exp(exp_or_name: run.Experiment | str): + if isinstance(exp_or_name, run.Experiment): + class _ExperimentCtx: + def __init__(self, exp): + self.exp = exp + + def __enter__(self): + return self.exp + + def __exit__(self, exc_type, exc, tb): + return False + + return _ExperimentCtx(exp_or_name) + + try: + return run.Experiment.from_title(exp_or_name) + except Exception: + return run.Experiment.from_id(exp_or_name) + + +def track_stage_tasks( + exp_or_name: run.Experiment | str, + cluster_config: Dict[str, Any], + *, + include_finished: bool = True, +) -> Dict[str, Any]: + """Track stage/task states through the configured execution backend.""" + backend = get_execution_backend(cluster_config) + with _with_exp(exp_or_name) as exp: + return backend.track_experiment(exp, include_finished=include_finished) + + +def stop_stage_tasks( + exp_or_name: run.Experiment | str, + cluster_config: Dict[str, Any], + *, + only_active: bool = True, +) -> list[str]: + """Stop stage tasks through the configured execution backend.""" + backend = get_execution_backend(cluster_config) + with _with_exp(exp_or_name) as exp: + return backend.stop_experiment(exp, only_active=only_active) diff --git a/nemo_skills/pipeline/utils/declarative.py b/nemo_skills/pipeline/utils/declarative.py index 53928e21bf..07eb02891f 100644 --- a/nemo_skills/pipeline/utils/declarative.py +++ b/nemo_skills/pipeline/utils/declarative.py @@ -31,9 +31,11 @@ run_exp, temporary_env_update, ) +from nemo_skills.pipeline.utils.backends import get_execution_backend from nemo_skills.pipeline.utils.exp import ( REUSE_CODE_EXP, get_packaging_job_key, + queue_ray_job_commands, tunnel_hash, ) from nemo_skills.pipeline.utils.mounts import get_mounts_from_config, is_mounted_filepath, normalize_mounts_list @@ -812,6 +814,7 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: shared_packager = None # Build commands and executors using prepared data + metadata_container_image: Optional[str] = None for entry_idx, entry in enumerate(prepared_commands): het_idx = entry["het_idx"] comp_idx = entry["comp_idx"] @@ -830,6 +833,8 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: # Resolve container and create executor container_image = self._resolve_container(exec_config, command, cluster_config) + if metadata_container_image is None: + metadata_container_image = container_image # Pass external dependencies only to the first executor in iteration order. # We use entry_idx rather than het_idx/comp_idx because prepared_commands may # have been reordered (e.g., to put spanning components first for allocation). @@ -902,10 +907,31 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: # Note: Path replacements for executor="none" are no longer needed with Script interface # Ray metadata handling - if self.with_ray and cluster_config["executor"] == "slurm": - metadata = {"use_with_ray_cluster": True} - else: - metadata = None + backend = get_execution_backend(cluster_config, with_ray=self.with_ray) + should_use_with_ray_cluster = bool(self.with_ray or getattr(backend, "name", "") == "ray") + metadata = backend.stage_metadata( + use_with_ray_cluster=should_use_with_ray_cluster, + container_image=metadata_container_image, + ) + + ray_queue_commands = [] + ray_queue_images = [] + for script, executor in zip(scripts, executors): + if not isinstance(script.inline, str): + continue + ray_queue_commands.append(script.inline) + ray_queue_images.append(getattr(executor, "container_image", None)) + + queue_ray_job_commands( + exp=exp, + backend=backend, + commands=ray_queue_commands, + command_images=ray_queue_images, + task_name=groups[0].name, + log_dir=log_dir, + task_dependencies=internal_deps, + should_use_with_ray_cluster=should_use_with_ray_cluster, + ) # Add to experiment and return task ID # Note: Internal dependencies (task handles from same experiment) go to exp.add() diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index 2eb2d8e4b2..a83bce1f7d 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -28,6 +28,7 @@ from nemo_run.core.execution.slurm import SlurmJobDetails, get_packaging_job_key from torchx.specs.api import AppState +from nemo_skills.pipeline.utils.backends import BackendRunOptions, get_execution_backend from nemo_skills.pipeline.utils.cluster import ( get_env_variables, get_slurm_timeout_str, @@ -241,6 +242,9 @@ def get_executor( Raised if a non-SLURM executor is requested with `num_nodes > 1`. """ env_vars = get_env_variables(cluster_config) + backend_env = get_execution_backend(cluster_config, with_ray=with_ray).get_env_overrides() + if backend_env: + env_vars.update(backend_env) config_mounts = get_mounts_from_config(cluster_config) if mounts is None: @@ -249,9 +253,23 @@ def get_executor( extra_package_dirs = tuple(extra_package_dirs) packager = get_packager(extra_package_dirs=extra_package_dirs) + # Ray backend with a precreated cluster allows multi-node without SLURM. + backend_config = cluster_config.get("backend") or cluster_config.get("execution_backend") or {} + if isinstance(backend_config, str): + backend_config = {"name": backend_config} + backend_name = str(backend_config.get("name", "")).strip().lower() + is_ray_precreated = backend_name == "ray" and bool(backend_config.get("precreated_cluster", False)) + if is_ray_precreated and not backend_config.get("endpoint"): + raise ValueError( + "Invalid cluster_config: backend.precreated_cluster=true requires backend.endpoint to be set." + ) + if cluster_config["executor"] != "slurm": - if num_nodes > 1: - raise ValueError("Local executor does not support multi-node execution") + if num_nodes > 1 and not is_ray_precreated: + raise ValueError( + "Local executor does not support multi-node execution. " + "Use executor: slurm or backend: ray with precreated_cluster: true" + ) if cluster_config["executor"] == "none": return LocalExecutor() @@ -675,6 +693,7 @@ def add_task( LOG.info("Adding a task with commands:") commands = [] + command_images = [] executors = [] def add_server_tasks(): @@ -720,6 +739,7 @@ def add_server_tasks(): if cluster_config["executor"] != "slurm" and num_server_tasks > 1: cmd_to_add = f"mpirun --allow-run-as-root -np {num_server_tasks} bash -c {shlex.quote(server_cmd)}" commands.append(cmd_to_add) + command_images.append(resolve_container_image(server_container, cluster_config)) executors.append(server_executor) het_group_indices.append(het_group) het_group += 1 @@ -757,6 +777,7 @@ def add_server_tasks(): with temporary_env_update(cluster_config, main_env_updates): cur_cmd = install_packages_wrap(cur_cmd, installation_command) commands.append(cur_cmd) + command_images.append(resolve_container_image(cur_container, cluster_config)) executors.append( get_executor( cluster_config=cluster_config, @@ -805,13 +826,15 @@ def add_server_tasks(): with temporary_env_update(cluster_config, sandbox_env_updates): commands.append(get_sandbox_command(cluster_config)) + sandbox_image = sandbox_container or cluster_config["containers"]["sandbox"] + command_images.append(resolve_container_image(sandbox_image, cluster_config)) if sandbox_mounts is not None: sandbox_exec_mounts = normalize_mounts_list(sandbox_mounts, allow_rw_mode=True) else: sandbox_exec_mounts = None if keep_mounts_for_sandbox else [] sandbox_executor = get_executor( cluster_config=cluster_config, - container=sandbox_container or cluster_config["containers"]["sandbox"], + container=sandbox_image, num_nodes=executors[0].nodes if cluster_config["executor"] == "slurm" else 1, tasks_per_node=1, gpus_per_node=0, @@ -890,10 +913,31 @@ def add_server_tasks(): ) commands[idx] = commands[idx].replace("/nemo_run/code", "./") - if with_ray and cluster_config["executor"] == "slurm": - metadata = {"use_with_ray_cluster": True} - else: - metadata = None + backend = get_execution_backend(cluster_config, with_ray=with_ray) + first_command_image = command_images[0] if command_images else None + should_use_with_ray_cluster = bool(with_ray or getattr(backend, "name", "") == "ray") + metadata = backend.stage_metadata( + use_with_ray_cluster=should_use_with_ray_cluster, + container_image=first_command_image, + ) + LOG.info( + "Execution backend resolved to '%s' (dashboard_url=%s).", + getattr(backend, "name", "unknown"), + getattr(backend, "dashboard_url", None), + ) + + # For Ray Jobs API mode, queue commands on the experiment and let the backend + # submit/track/cancel them centrally in start_experiment(). + queue_ray_job_commands( + exp=exp, + backend=backend, + commands=commands, + command_images=command_images, + task_name=task_name, + log_dir=log_dir, + task_dependencies=task_dependencies, + should_use_with_ray_cluster=should_use_with_ray_cluster, + ) if not task_dependencies: # empty list task_dependencies = None @@ -920,15 +964,66 @@ def add_server_tasks(): ) +def queue_ray_job_commands( + *, + exp, + backend, + commands: list[str], + command_images: list[str | None] | None, + task_name: str, + log_dir: str | None, + task_dependencies, + should_use_with_ray_cluster: bool, +) -> int: + """Queue commands for Ray Jobs API submission when the Ray backend is active.""" + if getattr(backend, "name", "") != "ray" or not getattr(backend, "dashboard_url", None): + return 0 + + queued_jobs = list(getattr(exp, "_ns_ray_jobs_queue", [])) + before_count = len(queued_jobs) + dep_names = [ + (dep if isinstance(dep, str) else getattr(dep, "name", str(dep))) + for dep in (task_dependencies or []) + ] + + for idx, command in enumerate(commands): + img = command_images[idx] if command_images and idx < len(command_images) else None + cmd_meta = backend.stage_metadata( + use_with_ray_cluster=should_use_with_ray_cluster, + container_image=img, + ) or {} + queued_task_name = task_name if len(commands) == 1 else f"{task_name}-{idx}" + selector = cmd_meta.get("entrypoint_label_selector") or cmd_meta.get("ray_entrypoint_label_selector") + job_log_file = f"{log_dir}/ray-jobs/{queued_task_name}.log" if log_dir else None + stage_manifest_file = f"{log_dir}/ray-jobs/manifest.jsonl" if log_dir else None + queued_jobs.append( + { + "task_name": queued_task_name, + "command": command, + "submission_id": queued_task_name, + "entrypoint_label_selector": selector, + "dep_task_names": dep_names, + "job_log_file": job_log_file, + "stage_run_id": task_name, + "stage_manifest_file": stage_manifest_file, + } + ) + + setattr(exp, "_ns_ray_jobs_queue", queued_jobs) + queued_count = len(queued_jobs) - before_count + LOG.info( + "Queued %d Ray Job command(s) for submission via %s.", + queued_count, + backend.dashboard_url, + ) + return queued_count + + def run_exp(exp, cluster_config, sequential=False, dry_run=False): """If sequential is not specified, using True locally and False otherwise. If it is specified, it will be used as is. """ - if dry_run: - LOG.info("Dry run mode is enabled, not running the experiment.") - return - if "mounts" in cluster_config: # Can only check cluster mounts here, not those added to add_task mounts = get_mounts_from_config(cluster_config) @@ -942,11 +1037,12 @@ def run_exp(exp, cluster_config, sequential=False, dry_run=False): ) check_remote_mount_directories(mount_sources, cluster_config, exit_on_failure=exit_if_failure) - if cluster_config["executor"] != "slurm": - exp.run(detach=False, tail_logs=True, sequential=sequential) - else: + backend = get_execution_backend(cluster_config) + run_options = BackendRunOptions(sequential=sequential, dry_run=dry_run) + + if cluster_config["executor"] == "slurm": try: - exp.run(detach=True, sequential=sequential) + backend.start_experiment(exp, cluster_config, run_options) except RuntimeError as e: if "Your repo has uncommitted changes." in str(e): raise RuntimeError( @@ -960,8 +1056,14 @@ def run_exp(exp, cluster_config, sequential=False, dry_run=False): ) else: raise + else: + backend.start_experiment(exp, cluster_config, run_options) - # caching the experiment code for reuse + if dry_run: + return + + # caching the experiment code for reuse + if cluster_config["executor"] == "slurm": tunnel = get_tunnel(cluster_config) cur_tunnel_hash = tunnel_hash(tunnel) if cur_tunnel_hash not in REUSE_CODE_EXP: diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py new file mode 100644 index 0000000000..404e265071 --- /dev/null +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -0,0 +1,824 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Ray execution backend for NeMo Skills pipeline. + +This module contains the RayBackend class responsible for submitting, +tracking, and stopping Ray Jobs via the Ray Jobs API +(dashboard_url / JobSubmissionClient). + +Key design points +----------------- +* Jobs are submitted **concurrently** using a ThreadPoolExecutor so that + paired workloads (e.g. GRPO training + vLLM judge server) can run at + the same time. The previous serial loop caused the judge to never + start while training was blocking, making the judge host-file handshake + time out. +* Each thread polls its own job until it reaches a terminal state, so + the overall start_experiment() call returns only after every submitted + job has finished (or one has failed). +* task_dependencies expressed as Ray Jobs are honoured by ordering + submission: a job whose deps are not yet SUCCEEDED is not submitted + until they are, using a ready-queue approach inside the concurrent + executor. +""" + +from __future__ import annotations + +import fnmatch +import json +import logging +import os +import re +import shlex +import threading +import time +import uuid +from concurrent.futures import Future, ThreadPoolExecutor, as_completed +from typing import Any, Dict + +import nemo_run as run + +from nemo_skills.pipeline.utils.backends import BackendRunOptions, ExecutionBackend +from nemo_skills.utils import get_logger_name + +LOG = logging.getLogger(get_logger_name(__file__)) + +# How long to sleep between Ray Jobs status polls (seconds). +_POLL_INTERVAL = 2 +_STOP_VERIFY_TIMEOUT = 30 + +# Terminal states that mean the job is done (success or failure). +_TERMINAL_STATES = {"SUCCEEDED", "FAILED", "STOPPED"} +_SUCCESS_STATE = "SUCCEEDED" +_FAILURE_STATES = {"FAILED", "STOPPED"} + + +class RayBackend(ExecutionBackend): + """Execution backend that submits tasks as Ray Jobs via the Jobs API. + + When ``dashboard_url`` is provided, jobs queued by ``add_task()`` + (stored on the experiment as ``_ns_ray_jobs_queue``) are submitted + concurrently to the Ray cluster through ``JobSubmissionClient``. + Pre-flight cluster checks are run before the first submission. + + Supports per-task ``entrypoint_label_selector`` so that tasks can be + routed to specific node pools based on framework labels (e.g. + ``framework=ray`` for CPU workers, ``framework=nemorl`` for GPU nodes). + """ + + name = "ray" + + def __init__( + self, + *, + endpoint: str | None = None, + precreated_cluster: bool = False, + control_plane: str | None = None, + kubernetes_mode: str | None = None, + dashboard_url: str | None = None, + entrypoint_label_selector: Dict[str, str] | None = None, + image_label_key: str | None = None, + image_label_selectors: Dict[str, Dict[str, str]] | None = None, + ): + self.endpoint = endpoint.strip() if endpoint else None + self.dashboard_url = self._normalize_dashboard_url(dashboard_url, self.endpoint) + self.precreated_cluster = precreated_cluster + self.control_plane = (control_plane or "").strip().lower() or None + if entrypoint_label_selector is not None and not isinstance(entrypoint_label_selector, dict): + raise ValueError("entrypoint_label_selector must be a dict[str, str] when provided") + self.entrypoint_label_selector = {str(k): str(v) for k, v in (entrypoint_label_selector or {}).items()} + self.image_label_key = (image_label_key or "").strip() or None + self.image_label_selectors = self._normalize_image_label_selectors(image_label_selectors) + self.kubernetes_mode = None + if self.control_plane == "kubernetes": + normalized_mode = (kubernetes_mode or "offline").strip().lower() + if normalized_mode not in {"offline", "online"}: + raise ValueError( + f"Unsupported kubernetes ray mode '{kubernetes_mode}'. " + "Supported values are 'offline' and 'online'." + ) + self.kubernetes_mode = normalized_mode + self._preflight_done = False + self._manifest_lock = threading.Lock() + + # ------------------------------------------------------------------ + # Helpers + # ------------------------------------------------------------------ + + @staticmethod + def _normalize_dashboard_url(dashboard_url: str | None, endpoint: str | None) -> str | None: + if dashboard_url: + return str(dashboard_url).strip() + if not endpoint: + return None + ep = str(endpoint).strip() + if ep.startswith("http://") or ep.startswith("https://"): + return ep + if ep.startswith("ray://"): + host_port = ep[len("ray://"):] + host = host_port.split(":", 1)[0] + return f"http://{host}:8265" + return None + + @staticmethod + def _sanitize_submission_id(value: str) -> str: + sanitized = re.sub(r"[^a-zA-Z0-9_.-]", "-", value) + return sanitized.strip("-._") or "job" + + @staticmethod + def _to_status_str(status: Any) -> str: + return str(getattr(status, "value", status)).upper() + + def _get_jobs_client(self): + if not self.dashboard_url: + raise RuntimeError( + "Ray Jobs submission requires backend.dashboard_url " + "(e.g. http://:8265)." + ) + try: + from ray.job_submission import JobSubmissionClient + except Exception as exc: + raise RuntimeError( + "Ray Jobs submission requires 'ray[jobs]' support in the submission environment." + ) from exc + return JobSubmissionClient(self.dashboard_url) + + def _stop_jobs_best_effort( + self, + client, + job_ids: list[str], + *, + timeout_s: int = _STOP_VERIFY_TIMEOUT, + reason: str = "cleanup", + ) -> None: + """Best-effort stop with bounded verification. + + Sends stop requests for all provided jobs, then polls status for up to + ``timeout_s`` seconds to confirm jobs reached a terminal state. This is + intentionally bounded to avoid perpetual waits during error handling. + """ + pending = {str(jid) for jid in job_ids if jid} + if not pending: + return + + for job_id in list(pending): + try: + client.stop_job(job_id) + except Exception as exc: + LOG.warning("Failed to send stop for Ray job %s during %s: %s", job_id, reason, exc) + + deadline = time.monotonic() + max(timeout_s, 0) + last_seen: Dict[str, str] = {} + while pending and time.monotonic() < deadline: + terminal_now: list[str] = [] + for job_id in list(pending): + try: + status = self._to_status_str(client.get_job_status(job_id)) + last_seen[job_id] = status + except Exception: + # Status lookups can transiently fail; keep trying until timeout. + continue + if status in _TERMINAL_STATES: + terminal_now.append(job_id) + for job_id in terminal_now: + pending.discard(job_id) + if pending: + time.sleep(_POLL_INTERVAL) + + if pending: + summary = {jid: last_seen.get(jid, "UNKNOWN") for jid in sorted(pending)} + LOG.warning( + "Timed out waiting for Ray job cleanup during %s after %ss; " + "jobs may still be running: %s", + reason, + timeout_s, + summary, + ) + + def _write_final_job_log( + self, + client, + *, + job_id: str, + task_name: str, + status: str, + job_log_file: str | None, + ) -> None: + """Persist final Ray job logs once the job reaches a terminal state. + + This is best-effort: failures to fetch/write logs are warned and ignored + so they do not mask the primary job outcome. + """ + if not job_log_file: + return + try: + logs = client.get_job_logs(job_id) or "" + except Exception as exc: + LOG.warning( + "Failed to fetch final logs for Ray job %s (%s): %s", + job_id, + task_name, + exc, + ) + return + + try: + os.makedirs(os.path.dirname(job_log_file), exist_ok=True) + with open(job_log_file, "w", encoding="utf-8") as f: + if logs: + f.write(logs) + if not logs.endswith("\n"): + f.write("\n") + f.write( + f"\n=== FINAL_STATUS task={task_name} job_id={job_id} " + f"status={status} ts={time.strftime('%Y-%m-%dT%H:%M:%S%z')} ===\n" + ) + except Exception as exc: + LOG.warning( + "Failed to write final logs for Ray job %s (%s) to %s: %s", + job_id, + task_name, + job_log_file, + exc, + ) + + def _append_stage_job_record( + self, + *, + stage_manifest_file: str | None, + record: Dict[str, Any], + ) -> None: + """Append one stage-job mapping record as JSONL (best-effort).""" + if not stage_manifest_file: + return + try: + os.makedirs(os.path.dirname(stage_manifest_file), exist_ok=True) + line = json.dumps(record, sort_keys=True) + with self._manifest_lock: + with open(stage_manifest_file, "a", encoding="utf-8") as f: + f.write(line) + f.write("\n") + except Exception as exc: + LOG.warning( + "Failed to append stage-job mapping record to %s: %s", + stage_manifest_file, + exc, + ) + + @staticmethod + def _prepare_job_entrypoint_command(command: str) -> str: + # When running *inside* a Ray Job, the driver is already on the cluster. + # Replace any forwarded Ray Client endpoint with local cluster autodiscovery. + stripped = re.sub(r"export\s+RAY_ADDRESS=[^&;]+&&\s*", "", command) + return f"export RAY_ADDRESS=auto && {stripped.strip()}" + + @staticmethod + def _normalize_image_label_selectors( + selectors: Dict[str, Dict[str, str]] | None, + ) -> Dict[str, Dict[str, str]]: + if selectors is None: + return {} + if not isinstance(selectors, dict): + raise ValueError( + "image_label_selectors must be a dict[str, dict[str, str]] when provided" + ) + normalized: Dict[str, Dict[str, str]] = {} + for pattern, labels in selectors.items(): + if not isinstance(labels, dict): + raise ValueError( + "image_label_selectors values must be dicts " + "(either label dict or {'key': ..., 'value': ...})" + ) + # Convenience form: {"key": "...", "value": "..."} + if set(labels.keys()) == {"key", "value"}: + normalized[str(pattern)] = {str(labels["key"]): str(labels["value"])} + else: + normalized[str(pattern)] = {str(k): str(v) for k, v in labels.items()} + return normalized + + def _labels_for_image(self, container_image: str) -> Dict[str, str]: + labels: Dict[str, str] = {} + for pattern, selector in self.image_label_selectors.items(): + if fnmatch.fnmatch(container_image, pattern): + for key, value in selector.items(): + labels.setdefault(key, value) + return labels + + @staticmethod + def _normalize_label_value(value: str) -> str: + normalized = re.sub(r"[^a-z0-9_.-]", "-", value.lower()) + normalized = re.sub(r"-+", "-", normalized).strip("-._") + return normalized or "unknown" + + # ------------------------------------------------------------------ + # Stage metadata + # ------------------------------------------------------------------ + + def stage_metadata( + self, + *, + use_with_ray_cluster: bool = False, + container_image: str | None = None, + ) -> Dict[str, Any] | None: + if self.precreated_cluster and self.endpoint: + should_use_embedded_ray_cluster = False + else: + should_use_embedded_ray_cluster = True + + metadata = super().stage_metadata(use_with_ray_cluster=should_use_embedded_ray_cluster) + metadata = dict(metadata or {}) + metadata["execution_backend"] = "kubernetes-ray" if self.control_plane == "kubernetes" else "ray" + if self.control_plane: + metadata["ray_control_plane"] = self.control_plane + if self.kubernetes_mode: + metadata["kubernetes_mode"] = self.kubernetes_mode + if self.endpoint: + metadata["ray_address"] = self.endpoint + metadata["ray_cluster_mode"] = "precreated" if self.precreated_cluster else "managed" + if self.dashboard_url: + metadata["ray_dashboard_url"] = self.dashboard_url + selector = dict(self.entrypoint_label_selector) + if container_image: + image_specific_selector = self._labels_for_image(container_image) + if image_specific_selector: + # Per-image/per-container selectors are intended to override + # defaults from entrypoint_label_selector when keys overlap. + selector.update(image_specific_selector) + elif self.image_label_key: + selector.setdefault( + self.image_label_key, self._normalize_label_value(container_image) + ) + if selector: + metadata["ray_entrypoint_label_selector"] = selector + metadata["entrypoint_label_selector"] = selector + return metadata + + def get_env_overrides(self) -> Dict[str, str]: + if not self.endpoint: + return {} + return {"RAY_ADDRESS": self.endpoint} + + # ------------------------------------------------------------------ + # Preflight + # ------------------------------------------------------------------ + + @staticmethod + def _preflight_config(cluster_config: Dict[str, Any]) -> Dict[str, Any]: + cfg = cluster_config.get("preflight") or {} + if cfg is None: + return {} + if not isinstance(cfg, dict): + raise ValueError("cluster_config.preflight must be a dict when provided") + return cfg + + @staticmethod + def _normalize_resource_key(key: str) -> str: + k = str(key).strip().lower() + if k in {"gpu", "gpus"}: + return "GPU" + if k in {"cpu", "cpus"}: + return "CPU" + if k in {"mem", "memory"}: + return "memory" + return str(key) + + @staticmethod + def _extract_node_labels(nodes_detail: list[Any]) -> list[Dict[str, str]]: + label_maps: list[Dict[str, str]] = [] + for node in nodes_detail: + if hasattr(node, "model_dump"): + node_data = node.model_dump() + elif hasattr(node, "dict"): + node_data = node.dict() + elif isinstance(node, dict): + node_data = node + else: + continue + labels = node_data.get("labels") or node_data.get("node_labels") or {} + if isinstance(labels, dict): + label_maps.append({str(k): str(v) for k, v in labels.items()}) + return label_maps + + def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOptions) -> None: + preflight_cfg = self._preflight_config(cluster_config) + enabled = bool(preflight_cfg.get("enabled", True)) + if not enabled: + return + + if self.precreated_cluster and not self.endpoint: + raise RuntimeError( + "Ray preflight failed: backend.precreated_cluster=true requires backend.endpoint." + ) + + if options.dry_run: + return + + if self._preflight_done: + return + + required_labels = preflight_cfg.get("required_node_labels") or [] + min_resources = preflight_cfg.get("min_cluster_resources") or {} + require_reachable = bool(preflight_cfg.get("require_ray_endpoint", self.precreated_cluster)) + strict_label_check = bool(preflight_cfg.get("strict_label_check", True)) + + if not require_reachable and not required_labels and not min_resources: + return + + if require_reachable and not self.endpoint: + raise RuntimeError( + "Ray preflight failed: backend.endpoint is required for connectivity checks." + ) + + try: + import ray + except Exception as exc: + raise RuntimeError( + "Ray preflight failed: python package 'ray' is required on the submission host." + ) from exc + + try: + init_kwargs = {"ignore_reinit_error": True, "logging_level": logging.ERROR} + if self.endpoint: + init_kwargs["address"] = self.endpoint + ray.init(**init_kwargs) + + live_nodes = [n for n in (ray.nodes() or []) if n.get("Alive", False)] + if require_reachable and not live_nodes: + raise RuntimeError( + "Ray preflight failed: endpoint reachable but no live nodes were found." + ) + + resources_total: Dict[str, float] = {} + for node in live_nodes: + for key, value in (node.get("Resources") or {}).items(): + try: + resources_total[str(key)] = resources_total.get(str(key), 0.0) + float(value) + except Exception: + continue + + if min_resources: + if not isinstance(min_resources, dict): + raise RuntimeError( + "Ray preflight failed: preflight.min_cluster_resources must be a dict." + ) + for key, required_value in min_resources.items(): + normalized_key = self._normalize_resource_key(str(key)) + try: + required_float = float(required_value) + except Exception as exc: + raise RuntimeError( + f"Ray preflight failed: invalid min_cluster_resources value " + f"for '{key}': {required_value}" + ) from exc + available = float(resources_total.get(normalized_key, 0.0)) + if available < required_float: + raise RuntimeError( + f"Ray preflight failed: resource '{normalized_key}' " + f"available={available} < required={required_float}." + ) + + if required_labels: + if not isinstance(required_labels, list): + raise RuntimeError( + "Ray preflight failed: preflight.required_node_labels must be a list." + ) + labels_by_node: list[Dict[str, str]] = [] + try: + from ray.util.state import list_nodes + labels_by_node = self._extract_node_labels(list_nodes(detail=True) or []) + except Exception: + labels_by_node = [] + + if not labels_by_node: + msg = ( + "Ray preflight could not inspect node labels from ray state API. " + "Set preflight.strict_label_check=false to bypass label checks." + ) + if strict_label_check: + raise RuntimeError(msg) + LOG.warning(msg) + else: + for label_req in required_labels: + if not isinstance(label_req, dict): + raise RuntimeError( + "Ray preflight failed: each required_node_labels item " + "must be a dict with key/value." + ) + key = str(label_req.get("key", "")).strip() + value = str(label_req.get("value", "")).strip() + if not key: + raise RuntimeError( + "Ray preflight failed: required_node_labels entries " + "must include non-empty key." + ) + if not any( + node_labels.get(key) == value for node_labels in labels_by_node + ): + raise RuntimeError( + f"Ray preflight failed: no live node matched label '{key}={value}'." + ) + + self._preflight_done = True + finally: + try: + ray.shutdown() + except Exception: + pass + + # ------------------------------------------------------------------ + # Experiment lifecycle + # ------------------------------------------------------------------ + + def start_experiment( + self, exp: run.Experiment, cluster_config: Dict[str, Any], options: BackendRunOptions + ): + self._run_preflight(cluster_config, options) + pending_jobs = getattr(exp, "_ns_ray_jobs_queue", None) + if not pending_jobs: + return super().start_experiment(exp, cluster_config, options) + + if options.dry_run: + LOG.info( + "Dry run mode enabled; skipping Ray Jobs submission for %d task(s).", + len(pending_jobs), + ) + return + + client = self._get_jobs_client() + self._submit_jobs_concurrently(client, exp, pending_jobs) + + def _submit_jobs_concurrently(self, client, exp: run.Experiment, pending_jobs: list) -> None: + """Submit all pending jobs concurrently so paired workloads (e.g. training + + judge) can run at the same time rather than serially. + + Dependency ordering + ------------------- + Each queued job may carry a ``dep_task_names`` list. A job is submitted + only after all its named dependencies have reached SUCCEEDED state. The + executor uses a simple ready-queue loop so dependencies are resolved + without blocking unrelated jobs. + + This fixes the finance GRPO training+judge pattern where: + - Training polls a host-file written by the judge + - The judge must therefore be running concurrently with training + - The old serial loop submitted training, blocked until it finished, + then submitted the judge — meaning the judge never came up while + training needed it, causing the wait-for-host-file loop to time out. + """ + # job_id -> status (populated as jobs complete) + completed: Dict[str, str] = {} + # job_id -> Future + futures: Dict[str, Future] = {} + # remaining jobs not yet submitted + pending = list(pending_jobs) + + exp_title = getattr(exp, "_title", "exp") + submitted_meta: list[Dict[str, Any]] = [] + + def _poll_until_done( + job_id: str, + task_name: str, + job_log_file: str | None, + stage_run_id: str | None, + stage_manifest_file: str | None, + submission_id: str | None, + ) -> Dict[str, Any]: + """Poll a single Ray Job until it reaches a terminal state.""" + last_logs = "" + while True: + status_str = self._to_status_str(client.get_job_status(job_id)) + try: + logs = client.get_job_logs(job_id) or "" + except Exception: + logs = "" + if logs and logs != last_logs: + delta = logs[len(last_logs):] if logs.startswith(last_logs) else logs + if delta.strip(): + for line in delta.rstrip().splitlines(): + LOG.info("ray-job/%s %s", job_id, line) + last_logs = logs + if status_str in _TERMINAL_STATES: + self._write_final_job_log( + client, + job_id=job_id, + task_name=task_name, + status=status_str, + job_log_file=job_log_file, + ) + self._append_stage_job_record( + stage_manifest_file=stage_manifest_file, + record={ + "event": "terminal", + "ts": time.strftime("%Y-%m-%dT%H:%M:%S%z"), + "stage_run_id": stage_run_id, + "task_name": task_name, + "job_id": job_id, + "submission_id": submission_id, + "status": status_str, + "dashboard_url": self.dashboard_url, + "job_log_file": job_log_file, + }, + ) + return {"job_id": job_id, "task_name": task_name, "status": status_str} + time.sleep(_POLL_INTERVAL) + + def _deps_satisfied(job: Dict[str, Any]) -> bool: + """Return True if all named dependencies have SUCCEEDED.""" + deps = job.get("dep_task_names") or [] + return all(completed.get(d) == _SUCCESS_STATE for d in deps) + + def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: + """Prepare and submit a single job; return its metadata.""" + command = str(job.get("command", "")).strip() + if not command: + raise ValueError(f"Ray job at index {idx} has an empty command.") + command = self._prepare_job_entrypoint_command(command) + entrypoint = f"bash -lc {shlex.quote(command)}" + selector = job.get("entrypoint_label_selector") + task_name = str(job.get("task_name", "nemo-run")) + metadata = {"nemo_task_name": task_name} + base_id = self._sanitize_submission_id( + str(job.get("submission_id") or f"{exp_title}-{idx}") + ) + submission_id = f"{base_id}-{uuid.uuid4().hex[:8]}" + LOG.info( + "Submitting Ray Job %s to %s (selector=%s)", + submission_id, + self.dashboard_url, + selector or {}, + ) + job_id = client.submit_job( + entrypoint=entrypoint, + submission_id=submission_id, + metadata=metadata, + entrypoint_label_selector=selector or None, + ) + return { + "task_name": task_name, + "job_id": job_id, + "submission_id": submission_id, + "job_log_file": job.get("job_log_file"), + "stage_run_id": job.get("stage_run_id"), + "stage_manifest_file": job.get("stage_manifest_file"), + } + + # Max workers = number of jobs so all can be in-flight simultaneously. + max_workers = max(len(pending_jobs), 1) + with ThreadPoolExecutor(max_workers=max_workers, thread_name_prefix="ray-job") as pool: + idx = 0 + while pending or futures: + # Submit any jobs whose dependencies are now satisfied. + still_pending = [] + for job in pending: + if _deps_satisfied(job): + try: + meta = _submit_one(job, idx) + except Exception as exc: + # Cancel all in-flight jobs on submission error. + self._stop_jobs_best_effort( + client, + list(futures.keys()), + reason="submission-failure", + ) + raise RuntimeError( + f"Failed to submit Ray job '{job.get('task_name', idx)}': {exc}" + ) from exc + idx += 1 + submitted_meta.append(meta) + self._append_stage_job_record( + stage_manifest_file=meta.get("stage_manifest_file"), + record={ + "event": "submitted", + "ts": time.strftime("%Y-%m-%dT%H:%M:%S%z"), + "stage_run_id": meta.get("stage_run_id"), + "task_name": meta.get("task_name"), + "job_id": meta.get("job_id"), + "submission_id": meta.get("submission_id"), + "status": "SUBMITTED", + "dashboard_url": self.dashboard_url, + "job_log_file": meta.get("job_log_file"), + }, + ) + future = pool.submit( + _poll_until_done, + meta["job_id"], + meta["task_name"], + meta.get("job_log_file"), + meta.get("stage_run_id"), + meta.get("stage_manifest_file"), + meta.get("submission_id"), + ) + futures[meta["job_id"]] = future + else: + still_pending.append(job) + pending = still_pending + + if not futures: + # No jobs in flight and no jobs ready — dependency deadlock. + if pending: + unresolved = [j.get("task_name", "?") for j in pending] + raise RuntimeError( + f"Ray Jobs dependency deadlock: tasks {unresolved} cannot be " + "satisfied (missing or failed dependency)." + ) + break + + # Wait for at least one future to finish before checking pending again. + done_iter = as_completed(futures.values(), timeout=None) + try: + done_future = next(done_iter) + except StopIteration: + break + + # Find the job_id for this future. + done_job_id = next( + (jid for jid, f in futures.items() if f is done_future), None + ) + if done_job_id is None: + continue + + result = done_future.result() # raises if _poll_until_done raised + status = result["status"] + task_name = result["task_name"] + completed[task_name] = status + del futures[done_job_id] + LOG.info("Ray job %s (%s) finished with status %s", done_job_id, task_name, status) + + if status in _FAILURE_STATES: + # Cancel all still-running jobs and fail fast. + self._stop_jobs_best_effort( + client, + list(futures.keys()), + reason=f"job-failure:{task_name}", + ) + raise RuntimeError( + f"Ray job '{task_name}' (id={done_job_id}) ended with status {status}." + ) + + setattr(exp, "_ns_ray_jobs_submitted", submitted_meta) + + def track_experiment( + self, exp: run.Experiment, include_finished: bool = True + ) -> Dict[str, Any]: + submitted = getattr(exp, "_ns_ray_jobs_submitted", None) + if not submitted or not self.dashboard_url: + return super().track_experiment(exp, include_finished=include_finished) + + client = self._get_jobs_client() + active_states = {"RUNNING", "PENDING", "SUBMITTED", "UNKNOWN"} + tracked: Dict[str, Any] = {} + for item in submitted: + task_name = str(item.get("task_name", "nemo-run")) + job_id = str(item.get("job_id", "")) + if not job_id: + continue + status = self._to_status_str(client.get_job_status(job_id)) + if include_finished or status in active_states: + tracked[task_name] = { + "status": status, + "handle": job_id, + "submission_id": item.get("submission_id"), + "stage_run_id": item.get("stage_run_id"), + "dashboard_url": self.dashboard_url, + "job_log_file": item.get("job_log_file"), + "stage_manifest_file": item.get("stage_manifest_file"), + } + return tracked + + def stop_experiment( + self, exp: run.Experiment, only_active: bool = True + ) -> list[str]: + submitted = getattr(exp, "_ns_ray_jobs_submitted", None) + if not submitted or not self.dashboard_url: + return super().stop_experiment(exp, only_active=only_active) + + client = self._get_jobs_client() + cancelled_jobs = [] + active_states = {"RUNNING", "PENDING", "SUBMITTED", "UNKNOWN"} + for item in submitted: + task_name = str(item.get("task_name", "nemo-run")) + job_id = str(item.get("job_id", "")) + if not job_id: + continue + status = self._to_status_str(client.get_job_status(job_id)) + if only_active and status not in active_states: + continue + self._stop_jobs_best_effort( + client, + [job_id], + reason=f"stop-experiment:{task_name}", + ) + cancelled_jobs.append(task_name) + return cancelled_jobs diff --git a/tests/test_backends.py b/tests/test_backends.py new file mode 100644 index 0000000000..1a3d140689 --- /dev/null +++ b/tests/test_backends.py @@ -0,0 +1,146 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from nemo_skills.pipeline.utils.backends import get_execution_backend + + +def test_kubernetes_ray_selector_with_image_label_translation(): + cluster_config = { + "executor": "slurm", + "backend": { + "name": "ray", + "control_plane": "kubernetes", + "precreated_cluster": True, + "endpoint": "ray://ray-head.ray.svc.cluster.local:10001", + "kubernetes": { + "mode": "offline", + "entrypoint_label_selector": {"type": "worker"}, + "image_label_key": "nemo/image", + }, + }, + } + + backend = get_execution_backend(cluster_config) + metadata = backend.stage_metadata(container_image="nvcr.io/nvidia/pytorch:25.02-py3") + + assert metadata["execution_backend"] == "kubernetes-ray" + assert metadata["ray_control_plane"] == "kubernetes" + assert metadata["ray_cluster_mode"] == "precreated" + assert metadata["kubernetes_mode"] == "offline" + assert metadata["entrypoint_label_selector"] == { + "type": "worker", + "nemo/image": "nvcr.io-nvidia-pytorch-25.02-py3", + } + assert metadata["ray_entrypoint_label_selector"] == metadata["entrypoint_label_selector"] + + +def test_kubernetes_ray_alias_uses_same_selector_logic(): + cluster_config = { + "executor": "slurm", + "backend": { + "name": "kubernetes-ray", + "endpoint": "ray://ray-head.ray.svc.cluster.local:10001", + "entrypoint_label_selector": {"type": "worker"}, + }, + } + + backend = get_execution_backend(cluster_config) + metadata = backend.stage_metadata(container_image="nvcr.io/nvidia/pytorch:25.02-py3") + + assert metadata["execution_backend"] == "kubernetes-ray" + assert metadata["entrypoint_label_selector"] == {"type": "worker"} + + +def test_image_translation_does_not_override_explicit_selector_value(): + cluster_config = { + "executor": "slurm", + "backend": { + "name": "ray", + "control_plane": "kubernetes", + "endpoint": "ray://ray-head.ray.svc.cluster.local:10001", + "entrypoint_label_selector": { + "type": "worker", + "nemo/image": "preset-image-label", + }, + "image_label_key": "nemo/image", + }, + } + + backend = get_execution_backend(cluster_config) + metadata = backend.stage_metadata(container_image="nvcr.io/nvidia/pytorch:25.02-py3") + + assert metadata["entrypoint_label_selector"]["nemo/image"] == "preset-image-label" + assert metadata["entrypoint_label_selector"]["type"] == "worker" + + +def test_image_label_selectors_add_explicit_key_value_pairs(): + cluster_config = { + "executor": "slurm", + "containers": { + "nemo-skills": "/containers/nemo-skills.sqsh", + "nemo-rl": "/containers/nemo-rl.sqsh", + }, + "backend": { + "name": "ray", + "control_plane": "kubernetes", + "endpoint": "ray://ray-head.ray.svc.cluster.local:10001", + "kubernetes": { + "mode": "offline", + "entrypoint_label_selector": {"type": "worker"}, + "image_label_selectors": { + "nemo-skills": { + "key": "nemo/workload", + "value": "skills", + }, + "nemo-rl": { + "nemo/workload": "rl", + }, + }, + }, + }, + } + + backend = get_execution_backend(cluster_config) + metadata = backend.stage_metadata(container_image="/containers/nemo-skills.sqsh") + + assert metadata["entrypoint_label_selector"]["type"] == "worker" + assert metadata["entrypoint_label_selector"]["nemo/workload"] == "skills" + + +def test_image_label_selectors_prefer_static_selector_keys(): + cluster_config = { + "executor": "slurm", + "containers": { + "nemo-skills": "/containers/nemo-skills.sqsh", + }, + "backend": { + "name": "ray", + "control_plane": "kubernetes", + "endpoint": "ray://ray-head.ray.svc.cluster.local:10001", + "kubernetes": { + "mode": "offline", + "entrypoint_label_selector": {"nemo/workload": "static"}, + "image_label_selectors": { + "nemo-skills": { + "nemo/workload": "dynamic", + } + }, + }, + }, + } + + backend = get_execution_backend(cluster_config) + metadata = backend.stage_metadata(container_image="/containers/nemo-skills.sqsh") + + assert metadata["entrypoint_label_selector"]["nemo/workload"] == "static" From 569cc153b4896d43fcb252dbbcc502f05884560b Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Mon, 1 Jun 2026 20:25:50 -0400 Subject: [PATCH 02/46] feat(backend): support the Ray Jobs backend in generate.py and cli.py generate.py: detect backend.name=ray and thread with_ray into the Pipeline; keep sequential mode off when running via the Ray backend. cli.py: guard the optional nemo_evaluator import so non-evaluator commands stay usable when its launcher dep is absent. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/cli.py | 8 +++++++- nemo_skills/pipeline/generate.py | 8 +++++++- 2 files changed, 14 insertions(+), 2 deletions(-) diff --git a/nemo_skills/pipeline/cli.py b/nemo_skills/pipeline/cli.py index d814dc8945..e83dffb2a7 100644 --- a/nemo_skills/pipeline/cli.py +++ b/nemo_skills/pipeline/cli.py @@ -25,7 +25,6 @@ from nemo_skills.pipeline.eval import eval from nemo_skills.pipeline.generate import generate from nemo_skills.pipeline.megatron_lm.train import train_megatron_lm -from nemo_skills.pipeline.nemo_evaluator import nemo_evaluator from nemo_skills.pipeline.nemo_gym_rollouts import nemo_gym_rollouts from nemo_skills.pipeline.nemo_rl.grpo import grpo_nemo_rl from nemo_skills.pipeline.nemo_rl.sft import sft_nemo_rl @@ -38,6 +37,13 @@ from nemo_skills.pipeline.summarize_robustness import summarize_robustness from nemo_skills.pipeline.verl.ppo import ppo_verl +try: + from nemo_skills.pipeline.nemo_evaluator import nemo_evaluator +except ModuleNotFoundError as e: + # Keep non-evaluator commands usable when optional launcher deps are absent. + if e.name != "nemo_evaluator_launcher": + raise + typer.main.get_command_name = lambda name: name diff --git a/nemo_skills/pipeline/generate.py b/nemo_skills/pipeline/generate.py index f1550da1b5..69f3110618 100644 --- a/nemo_skills/pipeline/generate.py +++ b/nemo_skills/pipeline/generate.py @@ -649,6 +649,11 @@ def convert_server_type_to_string(server_type): if not jobs: return None + backend_config = cluster_config.get("backend") or cluster_config.get("execution_backend") or {} + if isinstance(backend_config, str): + backend_config = {"name": backend_config} + with_ray_pipeline = str(backend_config.get("name", "")).strip().lower() == "ray" + # Create and run pipeline pipeline = Pipeline( name=expname, @@ -657,10 +662,11 @@ def convert_server_type_to_string(server_type): reuse_code=reuse_code, reuse_code_exp=reuse_code_exp, skip_hf_home_check=skip_hf_home_check, + with_ray=with_ray_pipeline, ) # TODO: remove after https://github.com/NVIDIA-NeMo/Skills/issues/578 is resolved as default will be single job - sequential = True if cluster_config["executor"] in ["local", "none"] else False + sequential = True if cluster_config["executor"] in ["local", "none"] and not with_ray_pipeline else False # Pass _reuse_exp to pipeline.run() to add jobs to existing experiment result = pipeline.run(dry_run=dry_run, _reuse_exp=_reuse_exp, sequential=sequential) From 0149074ae083f4bb8a40c971ecbd567c27d92910 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Wed, 3 Jun 2026 09:00:22 -0400 Subject: [PATCH 03/46] fix(ray-backend): explicit entrypoint_label_selector keys take precedence over image selectors The image_label_selectors branch used dict.update(), which overwrote keys that were explicitly set via entrypoint_label_selector with image-derived values. Switched to setdefault so explicitly-set static selector keys win over image-derived defaults, consistent with the image_label_key branch. Fixes test_image_label_selectors_prefer_static_selector_keys. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/ray_backend.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 404e265071..bd609c2b58 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -353,9 +353,10 @@ def stage_metadata( if container_image: image_specific_selector = self._labels_for_image(container_image) if image_specific_selector: - # Per-image/per-container selectors are intended to override - # defaults from entrypoint_label_selector when keys overlap. - selector.update(image_specific_selector) + # Image-derived selectors only fill gaps; explicit + # entrypoint_label_selector keys win (like image_label_key below). + for key, value in image_specific_selector.items(): + selector.setdefault(key, value) elif self.image_label_key: selector.setdefault( self.image_label_key, self._normalize_label_value(container_image) From 23e1038049f4d92801946b400fe59b8c96a6fe1e Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Mon, 8 Jun 2026 20:13:27 -0400 Subject: [PATCH 04/46] fix(ray-backend): resolve cross-experiment dependency deadlock When stages reuse one Experiment (exp.add(name="nemo-run")), a job's dependency names refer to nemo-run task handles, but completion was tracked only by task_name, so handle-named deps pointing at a prior/reused experiment were never satisfied and the resolver raised a false deadlock. Stamp each job's predicted nemo-run handle at queue time and record completion under both keys; treat deps on an already-finished prior experiment as satisfied (a genuinely failed dep still raises). Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/exp.py | 7 +++++ nemo_skills/pipeline/utils/ray_backend.py | 31 ++++++++++++++++++++--- 2 files changed, 35 insertions(+), 3 deletions(-) diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index a83bce1f7d..57db1afbb7 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -986,6 +986,12 @@ def queue_ray_job_commands( for dep in (task_dependencies or []) ] + # Predict the nemo-run handle exp.add() will assign this stage ("nemo-run", + # then "nemo-run_") so dep resolution can match deps named by handle. + base_handle_name = "nemo-run" + existing_jobs = len(getattr(exp, "jobs", []) or []) + task_handle = base_handle_name if existing_jobs == 0 else f"{base_handle_name}_{existing_jobs}" + for idx, command in enumerate(commands): img = command_images[idx] if command_images and idx < len(command_images) else None cmd_meta = backend.stage_metadata( @@ -1003,6 +1009,7 @@ def queue_ray_job_commands( "submission_id": queued_task_name, "entrypoint_label_selector": selector, "dep_task_names": dep_names, + "task_handle": task_handle, "job_log_file": job_log_file, "stage_run_id": task_name, "stage_manifest_file": stage_manifest_file, diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index bd609c2b58..cde47776fd 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -578,13 +578,22 @@ def _submit_jobs_concurrently(self, client, exp: run.Experiment, pending_jobs: l then submitted the judge — meaning the judge never came up while training needed it, causing the wait-for-host-file loop to time out. """ - # job_id -> status (populated as jobs complete) + # job_id -> status, keyed by both task_name and nemo-run handle + # so handle-named deps resolve against jobs in this batch. completed: Dict[str, str] = {} # job_id -> Future futures: Dict[str, Future] = {} # remaining jobs not yet submitted pending = list(pending_jobs) + # Deps naming a handle no job in this batch produces point at a prior + # reused experiment that already finished SUCCEEDED, so treat as satisfied. + producible_dep_names = { + j.get("task_handle") for j in pending_jobs if j.get("task_handle") + } + # job_id -> nemo-run task_handle (for recording completion under the handle name). + handle_by_job_id: Dict[str, str] = {} + exp_title = getattr(exp, "_title", "exp") submitted_meta: list[Dict[str, Any]] = [] @@ -636,9 +645,17 @@ def _poll_until_done( time.sleep(_POLL_INTERVAL) def _deps_satisfied(job: Dict[str, Any]) -> bool: - """Return True if all named dependencies have SUCCEEDED.""" + """True if every dep is SUCCEEDED, or names a handle this batch never + produces (a prior, already-SUCCEEDED experiment's job).""" deps = job.get("dep_task_names") or [] - return all(completed.get(d) == _SUCCESS_STATE for d in deps) + for d in deps: + if completed.get(d) == _SUCCESS_STATE: + continue + if d not in producible_dep_names: + # Dependency on a prior/reused-experiment job: already SUCCEEDED. + continue + return False + return True def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: """Prepare and submit a single job; return its metadata.""" @@ -668,6 +685,7 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: ) return { "task_name": task_name, + "task_handle": job.get("task_handle"), "job_id": job_id, "submission_id": submission_id, "job_log_file": job.get("job_log_file"), @@ -722,6 +740,8 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: meta.get("submission_id"), ) futures[meta["job_id"]] = future + if meta.get("task_handle"): + handle_by_job_id[meta["job_id"]] = meta["task_handle"] else: still_pending.append(job) pending = still_pending @@ -754,6 +774,11 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: status = result["status"] task_name = result["task_name"] completed[task_name] = status + # Also record completion under the nemo-run task_handle so that + # downstream jobs (whose dep_task_names are handles) can resolve. + done_handle = handle_by_job_id.get(done_job_id) + if done_handle: + completed[done_handle] = status del futures[done_job_id] LOG.info("Ray job %s (%s) finished with status %s", done_job_id, task_name, status) From 9c53e167d3e243d2da27062d85494a92e908eb25 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Mon, 8 Jun 2026 20:30:22 -0400 Subject: [PATCH 05/46] feat(ray-backend): forward cluster env vars to each Ray job's runtime_env Jobs were submitted with no runtime_env, so each inherited only the cluster head's environment; endpoint credentials configured for the cluster (NVIDIA_API_KEY / OPENAI_API_KEY for a judge, HF_TOKEN) never reached the job unless present in the head's env. Resolve the cluster env via get_env_variables() and pass it as runtime_env.env_vars on submit. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/backends.py | 10 +++++++- nemo_skills/pipeline/utils/ray_backend.py | 9 +++++++ tests/test_backends.py | 30 +++++++++++++++++++++++ 3 files changed, 48 insertions(+), 1 deletion(-) diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index 1fb0a829ae..8f87480468 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -20,6 +20,7 @@ import nemo_run as run +from nemo_skills.pipeline.utils.cluster import get_env_variables from nemo_skills.utils import get_logger_name LOG = logging.getLogger(get_logger_name(__file__)) @@ -175,7 +176,11 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa if not backend_name: if with_ray: - return RayBackend(endpoint=legacy_ray_endpoint, precreated_cluster=bool(legacy_ray_endpoint)) + return RayBackend( + endpoint=legacy_ray_endpoint, + precreated_cluster=bool(legacy_ray_endpoint), + env_vars=get_env_variables(cluster_config), + ) return ExecutionBackend() if backend_name in {"default", "none"}: @@ -213,6 +218,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa entrypoint_label_selector=selector, image_label_key=image_label_key, image_label_selectors=image_label_selectors, + env_vars=get_env_variables(cluster_config), ) return RayBackend( endpoint=endpoint, @@ -222,6 +228,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa entrypoint_label_selector=selector, image_label_key=image_label_key, image_label_selectors=image_label_selectors, + env_vars=get_env_variables(cluster_config), ) if backend_name in {"kubernetes-ray", "ray-kubernetes", "ray_kubernetes"}: k8s_cfg = backend_config.get("kubernetes") or {} @@ -252,6 +259,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa entrypoint_label_selector=selector, image_label_key=image_label_key, image_label_selectors=image_label_selectors, + env_vars=get_env_variables(cluster_config), ) raise ValueError( diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index cde47776fd..5b1f08126e 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -91,6 +91,7 @@ def __init__( entrypoint_label_selector: Dict[str, str] | None = None, image_label_key: str | None = None, image_label_selectors: Dict[str, Dict[str, str]] | None = None, + env_vars: Dict[str, str] | None = None, ): self.endpoint = endpoint.strip() if endpoint else None self.dashboard_url = self._normalize_dashboard_url(dashboard_url, self.endpoint) @@ -101,6 +102,9 @@ def __init__( self.entrypoint_label_selector = {str(k): str(v) for k, v in (entrypoint_label_selector or {}).items()} self.image_label_key = (image_label_key or "").strip() or None self.image_label_selectors = self._normalize_image_label_selectors(image_label_selectors) + # Forwarded to each job's runtime_env so cluster env vars (API keys, + # HF_TOKEN, ...) reach the job regardless of the cluster head's env. + self.env_vars = {str(k): str(v) for k, v in (env_vars or {}).items() if v is not None} self.kubernetes_mode = None if self.control_plane == "kubernetes": normalized_mode = (kubernetes_mode or "offline").strip().lower() @@ -155,6 +159,10 @@ def _get_jobs_client(self): ) from exc return JobSubmissionClient(self.dashboard_url) + def _build_runtime_env(self) -> Dict[str, Any] | None: + """runtime_env that forwards cluster env vars (API keys, HF_TOKEN, ...) to a job.""" + return {"env_vars": dict(self.env_vars)} if self.env_vars else None + def _stop_jobs_best_effort( self, client, @@ -682,6 +690,7 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: submission_id=submission_id, metadata=metadata, entrypoint_label_selector=selector or None, + runtime_env=self._build_runtime_env(), ) return { "task_name": task_name, diff --git a/tests/test_backends.py b/tests/test_backends.py index 1a3d140689..a2405314d3 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -144,3 +144,33 @@ def test_image_label_selectors_prefer_static_selector_keys(): metadata = backend.stage_metadata(container_image="/containers/nemo-skills.sqsh") assert metadata["entrypoint_label_selector"]["nemo/workload"] == "static" + + +def test_ray_backend_forwards_required_env_vars_to_runtime_env(): + cluster_config = { + "executor": "slurm", + "required_env_vars": ["MY_JUDGE_KEY=secret-123"], + "backend": {"name": "ray", "dashboard_url": "http://ray-head:8265"}, + } + + backend = get_execution_backend(cluster_config) + runtime_env = backend._build_runtime_env() + + assert runtime_env is not None + assert runtime_env["env_vars"]["MY_JUDGE_KEY"] == "secret-123" + + +def test_ray_backend_runtime_env_normalizes_and_filters_values(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + backend = RayBackend(dashboard_url="http://ray-head:8265", env_vars={"A": "1", "B": None, "C": 2}) + + assert backend._build_runtime_env() == {"env_vars": {"A": "1", "C": "2"}} + + +def test_ray_backend_runtime_env_none_when_no_env(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + backend = RayBackend(dashboard_url="http://ray-head:8265", env_vars={}) + + assert backend._build_runtime_env() is None From 3748a8a4e39fada7ea2c9818a6a4557249e886c6 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Mon, 8 Jun 2026 20:35:41 -0400 Subject: [PATCH 06/46] style: apply ruff format Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/backends.py | 18 ++--- nemo_skills/pipeline/utils/exp.py | 14 ++-- nemo_skills/pipeline/utils/ray_backend.py | 89 +++++++---------------- 3 files changed, 41 insertions(+), 80 deletions(-) diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index 8f87480468..54b20efef9 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -32,9 +32,7 @@ def _normalize_backend_config(cluster_config: Dict[str, Any]) -> Dict[str, Any]: if isinstance(backend_config, str): return {"name": backend_config} if not isinstance(backend_config, dict): - raise ValueError( - f"cluster_config backend must be a dict or string, got {type(backend_config).__name__}" - ) + raise ValueError(f"cluster_config backend must be a dict or string, got {type(backend_config).__name__}") return backend_config @@ -162,6 +160,7 @@ class _RayBackendShim(RayBackend): "stop_stage_tasks", ] + def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = False) -> ExecutionBackend: """Resolve execution backend from cluster config and compatibility flags. @@ -177,10 +176,10 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa if not backend_name: if with_ray: return RayBackend( - endpoint=legacy_ray_endpoint, - precreated_cluster=bool(legacy_ray_endpoint), - env_vars=get_env_variables(cluster_config), - ) + endpoint=legacy_ray_endpoint, + precreated_cluster=bool(legacy_ray_endpoint), + env_vars=get_env_variables(cluster_config), + ) return ExecutionBackend() if backend_name in {"default", "none"}: @@ -246,9 +245,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa or legacy_ray_endpoint ) dashboard_url = ( - backend_config.get("dashboard_url") - or backend_config.get("jobs_api_url") - or k8s_cfg.get("dashboard_url") + backend_config.get("dashboard_url") or backend_config.get("jobs_api_url") or k8s_cfg.get("dashboard_url") ) return RayBackend( endpoint=endpoint, @@ -269,6 +266,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa def _with_exp(exp_or_name: run.Experiment | str): if isinstance(exp_or_name, run.Experiment): + class _ExperimentCtx: def __init__(self, exp): self.exp = exp diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index 57db1afbb7..31dab0c38b 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -982,8 +982,7 @@ def queue_ray_job_commands( queued_jobs = list(getattr(exp, "_ns_ray_jobs_queue", [])) before_count = len(queued_jobs) dep_names = [ - (dep if isinstance(dep, str) else getattr(dep, "name", str(dep))) - for dep in (task_dependencies or []) + (dep if isinstance(dep, str) else getattr(dep, "name", str(dep))) for dep in (task_dependencies or []) ] # Predict the nemo-run handle exp.add() will assign this stage ("nemo-run", @@ -994,10 +993,13 @@ def queue_ray_job_commands( for idx, command in enumerate(commands): img = command_images[idx] if command_images and idx < len(command_images) else None - cmd_meta = backend.stage_metadata( - use_with_ray_cluster=should_use_with_ray_cluster, - container_image=img, - ) or {} + cmd_meta = ( + backend.stage_metadata( + use_with_ray_cluster=should_use_with_ray_cluster, + container_image=img, + ) + or {} + ) queued_task_name = task_name if len(commands) == 1 else f"{task_name}-{idx}" selector = cmd_meta.get("entrypoint_label_selector") or cmd_meta.get("ray_entrypoint_label_selector") job_log_file = f"{log_dir}/ray-jobs/{queued_task_name}.log" if log_dir else None diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 5b1f08126e..56b861995a 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -131,7 +131,7 @@ def _normalize_dashboard_url(dashboard_url: str | None, endpoint: str | None) -> if ep.startswith("http://") or ep.startswith("https://"): return ep if ep.startswith("ray://"): - host_port = ep[len("ray://"):] + host_port = ep[len("ray://") :] host = host_port.split(":", 1)[0] return f"http://{host}:8265" return None @@ -147,10 +147,7 @@ def _to_status_str(status: Any) -> str: def _get_jobs_client(self): if not self.dashboard_url: - raise RuntimeError( - "Ray Jobs submission requires backend.dashboard_url " - "(e.g. http://:8265)." - ) + raise RuntimeError("Ray Jobs submission requires backend.dashboard_url (e.g. http://:8265).") try: from ray.job_submission import JobSubmissionClient except Exception as exc: @@ -208,8 +205,7 @@ def _stop_jobs_best_effort( if pending: summary = {jid: last_seen.get(jid, "UNKNOWN") for jid in sorted(pending)} LOG.warning( - "Timed out waiting for Ray job cleanup during %s after %ss; " - "jobs may still be running: %s", + "Timed out waiting for Ray job cleanup during %s after %ss; jobs may still be running: %s", reason, timeout_s, summary, @@ -299,15 +295,12 @@ def _normalize_image_label_selectors( if selectors is None: return {} if not isinstance(selectors, dict): - raise ValueError( - "image_label_selectors must be a dict[str, dict[str, str]] when provided" - ) + raise ValueError("image_label_selectors must be a dict[str, dict[str, str]] when provided") normalized: Dict[str, Dict[str, str]] = {} for pattern, labels in selectors.items(): if not isinstance(labels, dict): raise ValueError( - "image_label_selectors values must be dicts " - "(either label dict or {'key': ..., 'value': ...})" + "image_label_selectors values must be dicts (either label dict or {'key': ..., 'value': ...})" ) # Convenience form: {"key": "...", "value": "..."} if set(labels.keys()) == {"key", "value"}: @@ -366,9 +359,7 @@ def stage_metadata( for key, value in image_specific_selector.items(): selector.setdefault(key, value) elif self.image_label_key: - selector.setdefault( - self.image_label_key, self._normalize_label_value(container_image) - ) + selector.setdefault(self.image_label_key, self._normalize_label_value(container_image)) if selector: metadata["ray_entrypoint_label_selector"] = selector metadata["entrypoint_label_selector"] = selector @@ -427,9 +418,7 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti return if self.precreated_cluster and not self.endpoint: - raise RuntimeError( - "Ray preflight failed: backend.precreated_cluster=true requires backend.endpoint." - ) + raise RuntimeError("Ray preflight failed: backend.precreated_cluster=true requires backend.endpoint.") if options.dry_run: return @@ -446,9 +435,7 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti return if require_reachable and not self.endpoint: - raise RuntimeError( - "Ray preflight failed: backend.endpoint is required for connectivity checks." - ) + raise RuntimeError("Ray preflight failed: backend.endpoint is required for connectivity checks.") try: import ray @@ -465,9 +452,7 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti live_nodes = [n for n in (ray.nodes() or []) if n.get("Alive", False)] if require_reachable and not live_nodes: - raise RuntimeError( - "Ray preflight failed: endpoint reachable but no live nodes were found." - ) + raise RuntimeError("Ray preflight failed: endpoint reachable but no live nodes were found.") resources_total: Dict[str, float] = {} for node in live_nodes: @@ -479,17 +464,14 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti if min_resources: if not isinstance(min_resources, dict): - raise RuntimeError( - "Ray preflight failed: preflight.min_cluster_resources must be a dict." - ) + raise RuntimeError("Ray preflight failed: preflight.min_cluster_resources must be a dict.") for key, required_value in min_resources.items(): normalized_key = self._normalize_resource_key(str(key)) try: required_float = float(required_value) except Exception as exc: raise RuntimeError( - f"Ray preflight failed: invalid min_cluster_resources value " - f"for '{key}': {required_value}" + f"Ray preflight failed: invalid min_cluster_resources value for '{key}': {required_value}" ) from exc available = float(resources_total.get(normalized_key, 0.0)) if available < required_float: @@ -500,12 +482,11 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti if required_labels: if not isinstance(required_labels, list): - raise RuntimeError( - "Ray preflight failed: preflight.required_node_labels must be a list." - ) + raise RuntimeError("Ray preflight failed: preflight.required_node_labels must be a list.") labels_by_node: list[Dict[str, str]] = [] try: from ray.util.state import list_nodes + labels_by_node = self._extract_node_labels(list_nodes(detail=True) or []) except Exception: labels_by_node = [] @@ -522,22 +503,16 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti for label_req in required_labels: if not isinstance(label_req, dict): raise RuntimeError( - "Ray preflight failed: each required_node_labels item " - "must be a dict with key/value." + "Ray preflight failed: each required_node_labels item must be a dict with key/value." ) key = str(label_req.get("key", "")).strip() value = str(label_req.get("value", "")).strip() if not key: raise RuntimeError( - "Ray preflight failed: required_node_labels entries " - "must include non-empty key." - ) - if not any( - node_labels.get(key) == value for node_labels in labels_by_node - ): - raise RuntimeError( - f"Ray preflight failed: no live node matched label '{key}={value}'." + "Ray preflight failed: required_node_labels entries must include non-empty key." ) + if not any(node_labels.get(key) == value for node_labels in labels_by_node): + raise RuntimeError(f"Ray preflight failed: no live node matched label '{key}={value}'.") self._preflight_done = True finally: @@ -550,9 +525,7 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti # Experiment lifecycle # ------------------------------------------------------------------ - def start_experiment( - self, exp: run.Experiment, cluster_config: Dict[str, Any], options: BackendRunOptions - ): + def start_experiment(self, exp: run.Experiment, cluster_config: Dict[str, Any], options: BackendRunOptions): self._run_preflight(cluster_config, options) pending_jobs = getattr(exp, "_ns_ray_jobs_queue", None) if not pending_jobs: @@ -596,9 +569,7 @@ def _submit_jobs_concurrently(self, client, exp: run.Experiment, pending_jobs: l # Deps naming a handle no job in this batch produces point at a prior # reused experiment that already finished SUCCEEDED, so treat as satisfied. - producible_dep_names = { - j.get("task_handle") for j in pending_jobs if j.get("task_handle") - } + producible_dep_names = {j.get("task_handle") for j in pending_jobs if j.get("task_handle")} # job_id -> nemo-run task_handle (for recording completion under the handle name). handle_by_job_id: Dict[str, str] = {} @@ -622,7 +593,7 @@ def _poll_until_done( except Exception: logs = "" if logs and logs != last_logs: - delta = logs[len(last_logs):] if logs.startswith(last_logs) else logs + delta = logs[len(last_logs) :] if logs.startswith(last_logs) else logs if delta.strip(): for line in delta.rstrip().splitlines(): LOG.info("ray-job/%s %s", job_id, line) @@ -675,9 +646,7 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: selector = job.get("entrypoint_label_selector") task_name = str(job.get("task_name", "nemo-run")) metadata = {"nemo_task_name": task_name} - base_id = self._sanitize_submission_id( - str(job.get("submission_id") or f"{exp_title}-{idx}") - ) + base_id = self._sanitize_submission_id(str(job.get("submission_id") or f"{exp_title}-{idx}")) submission_id = f"{base_id}-{uuid.uuid4().hex[:8]}" LOG.info( "Submitting Ray Job %s to %s (selector=%s)", @@ -773,9 +742,7 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: break # Find the job_id for this future. - done_job_id = next( - (jid for jid, f in futures.items() if f is done_future), None - ) + done_job_id = next((jid for jid, f in futures.items() if f is done_future), None) if done_job_id is None: continue @@ -798,15 +765,11 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: list(futures.keys()), reason=f"job-failure:{task_name}", ) - raise RuntimeError( - f"Ray job '{task_name}' (id={done_job_id}) ended with status {status}." - ) + raise RuntimeError(f"Ray job '{task_name}' (id={done_job_id}) ended with status {status}.") setattr(exp, "_ns_ray_jobs_submitted", submitted_meta) - def track_experiment( - self, exp: run.Experiment, include_finished: bool = True - ) -> Dict[str, Any]: + def track_experiment(self, exp: run.Experiment, include_finished: bool = True) -> Dict[str, Any]: submitted = getattr(exp, "_ns_ray_jobs_submitted", None) if not submitted or not self.dashboard_url: return super().track_experiment(exp, include_finished=include_finished) @@ -832,9 +795,7 @@ def track_experiment( } return tracked - def stop_experiment( - self, exp: run.Experiment, only_active: bool = True - ) -> list[str]: + def stop_experiment(self, exp: run.Experiment, only_active: bool = True) -> list[str]: submitted = getattr(exp, "_ns_ray_jobs_submitted", None) if not submitted or not self.dashboard_url: return super().stop_experiment(exp, only_active=only_active) From b6cb8225fcb1b2b7de16c16f3dadbcdca555033b Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 9 Jun 2026 14:24:07 -0400 Subject: [PATCH 07/46] docs: note NVFlow 2-cluster routing in the precreated-Ray sample config Signed-off-by: Nick Gupta --- cluster_configs/example-slurm-ray-k8s-precreated.yaml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/cluster_configs/example-slurm-ray-k8s-precreated.yaml b/cluster_configs/example-slurm-ray-k8s-precreated.yaml index 6a3e4cfae5..1ba510970c 100644 --- a/cluster_configs/example-slurm-ray-k8s-precreated.yaml +++ b/cluster_configs/example-slurm-ray-k8s-precreated.yaml @@ -56,6 +56,9 @@ timeouts: CHANGE_ME_CPU_PARTITION: "1-00:00:00" # Backend config: pre-created Ray cluster on Kubernetes. +# This sample targets ONE cluster. For NVFlow's 2-cluster (CPU + GPU) setup that +# auto-routes each stage by its num_gpus, see NVFlow's +# docs/recipes/finance/install-ray.md (backend.dashboard_url + cpu_dashboard_url). backend: name: ray control_plane: kubernetes From c758108663f9f002cada37f28b5c4885e108bd8c Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 9 Jun 2026 15:40:49 -0400 Subject: [PATCH 08/46] fix(ray-backend): forward external deps and require proven completion Cross-experiment dependencies (resolved run_after handles) were dropped when queuing Ray Jobs, so a job could submit before a prerequisite in another experiment finished. Forward them alongside same-experiment task_dependencies from both the add_task and declarative pipeline paths. Tighten the queue's dependency resolver so a dependency produced by a job in the current batch is satisfied only once that job reaches SUCCEEDED (tracked by both task_name and nemo-run handle), preventing a dependent from starting early. A dependency that matches no job in the batch references a prior or cross-experiment job already gated by the prerequisite experiment's blocking submission, so it stays treated as satisfied and the resolver does not deadlock on a job it cannot observe. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/declarative.py | 5 +++ nemo_skills/pipeline/utils/exp.py | 22 ++++++++--- nemo_skills/pipeline/utils/ray_backend.py | 45 ++++++++++++++++++----- 3 files changed, 57 insertions(+), 15 deletions(-) diff --git a/nemo_skills/pipeline/utils/declarative.py b/nemo_skills/pipeline/utils/declarative.py index 07eb02891f..4ce41b5bbb 100644 --- a/nemo_skills/pipeline/utils/declarative.py +++ b/nemo_skills/pipeline/utils/declarative.py @@ -922,6 +922,10 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: ray_queue_commands.append(script.inline) ray_queue_images.append(getattr(executor, "container_image", None)) + # Forward both internal (same-experiment) and external (cross-experiment + # run_after) dependencies. Dropping external deps would let Ray jobs submit + # before their prerequisites finish, since Ray ordering is resolved from + # the queued dep names rather than the nemo-run executor. queue_ray_job_commands( exp=exp, backend=backend, @@ -930,6 +934,7 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: task_name=groups[0].name, log_dir=log_dir, task_dependencies=internal_deps, + external_dependencies=external_deps, should_use_with_ray_cluster=should_use_with_ray_cluster, ) diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index 31dab0c38b..fab5e11c9b 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -927,7 +927,9 @@ def add_server_tasks(): ) # For Ray Jobs API mode, queue commands on the experiment and let the backend - # submit/track/cancel them centrally in start_experiment(). + # submit/track/cancel them centrally in start_experiment(). `dependencies` + # holds the resolved external run_after handles; forward them too so the Ray + # queue waits on cross-experiment prerequisites, not just same-experiment ones. queue_ray_job_commands( exp=exp, backend=backend, @@ -936,6 +938,7 @@ def add_server_tasks(): task_name=task_name, log_dir=log_dir, task_dependencies=task_dependencies, + external_dependencies=dependencies, should_use_with_ray_cluster=should_use_with_ray_cluster, ) @@ -974,16 +977,25 @@ def queue_ray_job_commands( log_dir: str | None, task_dependencies, should_use_with_ray_cluster: bool, + external_dependencies=None, ) -> int: - """Queue commands for Ray Jobs API submission when the Ray backend is active.""" + """Queue commands for Ray Jobs API submission when the Ray backend is active. + + Both within-experiment ``task_dependencies`` and cross-experiment + ``external_dependencies`` (resolved ``run_after`` handles) are forwarded so + the Ray queue can order submissions on all declared prerequisites. + """ if getattr(backend, "name", "") != "ray" or not getattr(backend, "dashboard_url", None): return 0 queued_jobs = list(getattr(exp, "_ns_ray_jobs_queue", [])) before_count = len(queued_jobs) - dep_names = [ - (dep if isinstance(dep, str) else getattr(dep, "name", str(dep))) for dep in (task_dependencies or []) - ] + + def _dep_name(dep): + return dep if isinstance(dep, str) else getattr(dep, "name", str(dep)) + + all_deps = list(task_dependencies or []) + list(external_dependencies or []) + dep_names = [_dep_name(dep) for dep in all_deps] # Predict the nemo-run handle exp.add() will assign this stage ("nemo-run", # then "nemo-run_") so dep resolution can match deps named by handle. diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 56b861995a..a7fe7b9681 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -567,9 +567,15 @@ def _submit_jobs_concurrently(self, client, exp: run.Experiment, pending_jobs: l # remaining jobs not yet submitted pending = list(pending_jobs) - # Deps naming a handle no job in this batch produces point at a prior - # reused experiment that already finished SUCCEEDED, so treat as satisfied. - producible_dep_names = {j.get("task_handle") for j in pending_jobs if j.get("task_handle")} + # Dependency identifiers this batch can observe to completion: the nemo-run + # handle and the task_name of every queued job. A dep matching one of these + # MUST reach SUCCEEDED before its dependents start (it is in flight here). A + # dep matching none of them references a prior or cross-experiment job that + # is gated upstream (the prerequisite experiment's blocking start_experiment + # already returned before this one began), so it is treated as satisfied to + # avoid a cross-experiment deadlock the resolver cannot otherwise break. + in_batch_dep_names = {j.get("task_handle") for j in pending_jobs if j.get("task_handle")} + in_batch_dep_names |= {j.get("task_name") for j in pending_jobs if j.get("task_name")} # job_id -> nemo-run task_handle (for recording completion under the handle name). handle_by_job_id: Dict[str, str] = {} @@ -624,16 +630,24 @@ def _poll_until_done( time.sleep(_POLL_INTERVAL) def _deps_satisfied(job: Dict[str, Any]) -> bool: - """True if every dep is SUCCEEDED, or names a handle this batch never - produces (a prior, already-SUCCEEDED experiment's job).""" + """True only when every dependency is provably satisfied. + + A dependency on a job submitted in this batch is satisfied only once + that job has reached SUCCEEDED (recorded under its task_name and its + nemo-run handle). It is never assumed done while still in flight, so a + dependent cannot start prematurely. A dependency that matches no job in + this batch is a prior or cross-experiment job already gated upstream and + is treated as satisfied so the resolver does not deadlock waiting on a + job it can never observe. + """ deps = job.get("dep_task_names") or [] for d in deps: if completed.get(d) == _SUCCESS_STATE: continue - if d not in producible_dep_names: - # Dependency on a prior/reused-experiment job: already SUCCEEDED. - continue - return False + if d in in_batch_dep_names: + # In flight in this batch and not yet SUCCEEDED: keep waiting. + return False + # Not produced by this batch: gated upstream; treat as satisfied. return True def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: @@ -746,7 +760,18 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: if done_job_id is None: continue - result = done_future.result() # raises if _poll_until_done raised + try: + result = done_future.result() # raises if _poll_until_done raised + except Exception as exc: + # Polling the Jobs API failed; stop every still-submitted job + # (including the one whose poll failed, still in `futures`) so we + # do not leave orphaned jobs running on the cluster, then re-raise. + self._stop_jobs_best_effort( + client, + list(futures.keys()), + reason="poll-failure", + ) + raise RuntimeError(f"Ray Jobs polling failed; submitted jobs were stopped: {exc}") from exc status = result["status"] task_name = result["task_name"] completed[task_name] = status From 9408b56b3baa3ccd5730885dd529d4a3b99f9187 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 9 Jun 2026 15:40:56 -0400 Subject: [PATCH 09/46] fix(backend): treat dashboard_url as precreated signal; narrow exp lookup Resolve precreated_cluster as the OR of an explicit precreated_cluster flag, a Ray Client endpoint, and a dashboard_url, and pass that into every RayBackend construction (the ray branch, its kubernetes sub-branch, and the kubernetes-ray alias). The dashboard-only path now stages as a precreated cluster instead of requesting an embedded one, and an explicit precreated_cluster: true is no longer downgraded. Narrow the experiment-lookup fallback to catch only FileNotFoundError from Experiment.from_title so unrelated errors propagate instead of silently resolving by id, matching get_exp_handles. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/backends.py | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index 54b20efef9..d908ebacc9 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -196,7 +196,13 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa or backend_config.get("jobs_api_url") or (backend_config.get("kubernetes") or {}).get("dashboard_url") ) - precreated_cluster = bool(backend_config.get("precreated_cluster", False)) or bool(endpoint) + # A dashboard_url (Jobs API) targets an existing cluster just like an + # explicit precreated_cluster flag or a Ray Client endpoint. Treat any of + # them as the precreated signal, and never downgrade an explicit + # precreated_cluster: true. + precreated_cluster = ( + bool(backend_config.get("precreated_cluster", False)) or bool(endpoint) or bool(dashboard_url) + ) control_plane = str(backend_config.get("control_plane") or "").strip().lower() k8s_cfg = backend_config.get("kubernetes") or {} selector = backend_config.get("entrypoint_label_selector") or k8s_cfg.get("entrypoint_label_selector") @@ -210,7 +216,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa endpoint = endpoint or k8s_cfg.get("endpoint") return RayBackend( endpoint=endpoint, - precreated_cluster=bool(endpoint), + precreated_cluster=precreated_cluster, control_plane="kubernetes", kubernetes_mode=mode, dashboard_url=dashboard_url, @@ -247,9 +253,13 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa dashboard_url = ( backend_config.get("dashboard_url") or backend_config.get("jobs_api_url") or k8s_cfg.get("dashboard_url") ) + # endpoint, dashboard_url, or an explicit flag all signal a precreated cluster. + precreated_cluster = ( + bool(backend_config.get("precreated_cluster", False)) or bool(endpoint) or bool(dashboard_url) + ) return RayBackend( endpoint=endpoint, - precreated_cluster=bool(endpoint), + precreated_cluster=precreated_cluster, control_plane="kubernetes", kubernetes_mode=mode, dashboard_url=dashboard_url, @@ -281,7 +291,9 @@ def __exit__(self, exc_type, exc, tb): try: return run.Experiment.from_title(exp_or_name) - except Exception: + except FileNotFoundError: + # Title not found; fall back to treating the argument as an experiment id. + # Any other error from from_title is a real failure and must propagate. return run.Experiment.from_id(exp_or_name) From c0aa065d32d247f7b91c2777169306bfced3e842 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 9 Jun 2026 15:41:31 -0400 Subject: [PATCH 10/46] docs(config): move sample preflight block to top-level config scope The commented preflight example was nested under backend:, but preflight is read from the top level of the cluster config. Dedent it to a sibling of backend: so uncommenting it takes effect, and note the scope inline. Signed-off-by: Nick Gupta --- .../example-slurm-ray-k8s-precreated.yaml | 41 ++++++++++--------- 1 file changed, 21 insertions(+), 20 deletions(-) diff --git a/cluster_configs/example-slurm-ray-k8s-precreated.yaml b/cluster_configs/example-slurm-ray-k8s-precreated.yaml index 1ba510970c..a305c06f77 100644 --- a/cluster_configs/example-slurm-ray-k8s-precreated.yaml +++ b/cluster_configs/example-slurm-ray-k8s-precreated.yaml @@ -84,26 +84,27 @@ backend: # key: nemo/workload # value: rl - # Optional preflight checks run before Ray-backed experiment start. - # These are useful for failing fast when endpoint, labels, or cluster capacity - # are not aligned with the job requirements. - # preflight: - # enabled: true - # # If true, require Ray endpoint connectivity and at least one live node. - # require_ray_endpoint: true - # # If true, fail if label inspection cannot be performed. - # strict_label_check: true - # # Require at least one live node matching each key/value pair. - # required_node_labels: - # - key: nemo/has-nemo-rl - # value: "true" - # - key: workload - # value: "training" - # # Require minimum total live-cluster resources before submission. - # # Supported aliases: gpu|gpus -> GPU, cpu|cpus -> CPU, mem|memory -> memory. - # min_cluster_resources: - # gpu: 16 - # cpu: 64 +# Optional preflight checks run before Ray-backed experiment start. +# Read from the TOP LEVEL of the cluster config (a sibling of `backend:`), not +# from under `backend:`. Useful for failing fast when endpoint, labels, or +# cluster capacity are not aligned with the job requirements. +# preflight: +# enabled: true +# # If true, require Ray endpoint connectivity and at least one live node. +# require_ray_endpoint: true +# # If true, fail if label inspection cannot be performed. +# strict_label_check: true +# # Require at least one live node matching each key/value pair. +# required_node_labels: +# - key: nemo/has-nemo-rl +# value: "true" +# - key: workload +# value: "training" +# # Require minimum total live-cluster resources before submission. +# # Supported aliases: gpu|gpus -> GPU, cpu|cpus -> CPU, mem|memory -> memory. +# min_cluster_resources: +# gpu: 16 +# cpu: 64 # Required mounts for models/data/workspace. mounts: From 2c596b5c3ae2d8b3d04e2bb0a6cb2d7d59a7fac4 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 9 Jun 2026 15:56:45 -0400 Subject: [PATCH 11/46] test(ray-backend): cover dependency resolver and poll-failure cleanup Extract the in-batch dependency-name set and the dependency predicate into pure static methods (_compute_in_batch_dep_names, _deps_satisfied) and the poll-failure cleanup into _handle_poll_failure, then call them from _submit_jobs_concurrently. Behavior is unchanged; this only makes the logic importable for unit tests. Add tests covering: the dep-name set includes both handles and task_names and ignores jobs missing them; an in-batch dependency blocks dependents until it reaches SUCCEEDED (under task_name or handle); a dependency matching no job in the batch is treated as satisfied; mixed in-batch/cross-batch deps stay blocked until the in-batch one succeeds; empty/none deps are satisfied; poll-failure cleanup stops submitted jobs once and chains the original error; and best-effort stop keeps going when one stop call raises. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/ray_backend.py | 96 +++++++++-------- tests/test_backends.py | 120 ++++++++++++++++++++++ 2 files changed, 175 insertions(+), 41 deletions(-) diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index a7fe7b9681..1d3196c443 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -46,7 +46,7 @@ import time import uuid from concurrent.futures import Future, ThreadPoolExecutor, as_completed -from typing import Any, Dict +from typing import Any, Dict, NoReturn import nemo_run as run @@ -145,6 +145,44 @@ def _sanitize_submission_id(value: str) -> str: def _to_status_str(status: Any) -> str: return str(getattr(status, "value", status)).upper() + @staticmethod + def _compute_in_batch_dep_names(jobs: list[Dict[str, Any]]) -> set[str]: + """Dependency identifiers a batch can observe to completion. + + The set is the nemo-run handle and the task_name of every queued job. + Jobs missing those keys contribute nothing. A dep matching this set is in + flight in the batch and must reach SUCCEEDED before its dependents start; + a dep matching none of it is gated upstream (see ``_deps_satisfied``). + """ + names = {j.get("task_handle") for j in jobs if j.get("task_handle")} + names |= {j.get("task_name") for j in jobs if j.get("task_name")} + return names + + @staticmethod + def _deps_satisfied( + dep_names: list[str] | None, + completed: Dict[str, str], + in_batch_dep_names: set[str], + ) -> bool: + """True only when every dependency is provably satisfied. + + A dependency on a job submitted in this batch is satisfied only once that + job has reached SUCCEEDED (recorded under its task_name and its nemo-run + handle). It is never assumed done while still in flight, so a dependent + cannot start prematurely. A dependency that matches no job in this batch is + a prior or cross-experiment job already gated upstream and is treated as + satisfied so the resolver does not deadlock waiting on a job it can never + observe. + """ + for d in dep_names or []: + if completed.get(d) == _SUCCESS_STATE: + continue + if d in in_batch_dep_names: + # In flight in this batch and not yet SUCCEEDED: keep waiting. + return False + # Not produced by this batch: gated upstream; treat as satisfied. + return True + def _get_jobs_client(self): if not self.dashboard_url: raise RuntimeError("Ray Jobs submission requires backend.dashboard_url (e.g. http://:8265).") @@ -211,6 +249,16 @@ def _stop_jobs_best_effort( summary, ) + def _handle_poll_failure(self, client, job_ids: list[str], exc: Exception) -> NoReturn: + """Stop every still-submitted job, then re-raise the polling error. + + Called when polling the Jobs API fails; stops all jobs still in flight + (including the one whose poll failed) so none are orphaned on the cluster, + then raises a wrapped RuntimeError chained from the original exception. + """ + self._stop_jobs_best_effort(client, job_ids, reason="poll-failure") + raise RuntimeError(f"Ray Jobs polling failed; submitted jobs were stopped: {exc}") from exc + def _write_final_job_log( self, client, @@ -567,15 +615,10 @@ def _submit_jobs_concurrently(self, client, exp: run.Experiment, pending_jobs: l # remaining jobs not yet submitted pending = list(pending_jobs) - # Dependency identifiers this batch can observe to completion: the nemo-run - # handle and the task_name of every queued job. A dep matching one of these - # MUST reach SUCCEEDED before its dependents start (it is in flight here). A - # dep matching none of them references a prior or cross-experiment job that - # is gated upstream (the prerequisite experiment's blocking start_experiment - # already returned before this one began), so it is treated as satisfied to - # avoid a cross-experiment deadlock the resolver cannot otherwise break. - in_batch_dep_names = {j.get("task_handle") for j in pending_jobs if j.get("task_handle")} - in_batch_dep_names |= {j.get("task_name") for j in pending_jobs if j.get("task_name")} + # Dependency identifiers this batch can observe to completion (handles and + # task_names of every queued job). A dep matching none of them is gated + # upstream and treated as satisfied; see _deps_satisfied / its helpers. + in_batch_dep_names = self._compute_in_batch_dep_names(pending_jobs) # job_id -> nemo-run task_handle (for recording completion under the handle name). handle_by_job_id: Dict[str, str] = {} @@ -629,27 +672,6 @@ def _poll_until_done( return {"job_id": job_id, "task_name": task_name, "status": status_str} time.sleep(_POLL_INTERVAL) - def _deps_satisfied(job: Dict[str, Any]) -> bool: - """True only when every dependency is provably satisfied. - - A dependency on a job submitted in this batch is satisfied only once - that job has reached SUCCEEDED (recorded under its task_name and its - nemo-run handle). It is never assumed done while still in flight, so a - dependent cannot start prematurely. A dependency that matches no job in - this batch is a prior or cross-experiment job already gated upstream and - is treated as satisfied so the resolver does not deadlock waiting on a - job it can never observe. - """ - deps = job.get("dep_task_names") or [] - for d in deps: - if completed.get(d) == _SUCCESS_STATE: - continue - if d in in_batch_dep_names: - # In flight in this batch and not yet SUCCEEDED: keep waiting. - return False - # Not produced by this batch: gated upstream; treat as satisfied. - return True - def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: """Prepare and submit a single job; return its metadata.""" command = str(job.get("command", "")).strip() @@ -693,7 +715,7 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: # Submit any jobs whose dependencies are now satisfied. still_pending = [] for job in pending: - if _deps_satisfied(job): + if self._deps_satisfied(job.get("dep_task_names"), completed, in_batch_dep_names): try: meta = _submit_one(job, idx) except Exception as exc: @@ -763,15 +785,7 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: try: result = done_future.result() # raises if _poll_until_done raised except Exception as exc: - # Polling the Jobs API failed; stop every still-submitted job - # (including the one whose poll failed, still in `futures`) so we - # do not leave orphaned jobs running on the cluster, then re-raise. - self._stop_jobs_best_effort( - client, - list(futures.keys()), - reason="poll-failure", - ) - raise RuntimeError(f"Ray Jobs polling failed; submitted jobs were stopped: {exc}") from exc + self._handle_poll_failure(client, list(futures.keys()), exc) status = result["status"] task_name = result["task_name"] completed[task_name] = status diff --git a/tests/test_backends.py b/tests/test_backends.py index a2405314d3..85e76ee368 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -12,6 +12,8 @@ # See the License for the specific language governing permissions and # limitations under the License. +import pytest + from nemo_skills.pipeline.utils.backends import get_execution_backend @@ -174,3 +176,121 @@ def test_ray_backend_runtime_env_none_when_no_env(): backend = RayBackend(dashboard_url="http://ray-head:8265", env_vars={}) assert backend._build_runtime_env() is None + + +# --------------------------------------------------------------------------- +# Dependency resolver (in-batch vs cross-experiment deps) +# --------------------------------------------------------------------------- + + +def test_compute_in_batch_dep_names_includes_handles_and_task_names(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + jobs = [ + {"task_name": "train", "task_handle": "train-handle"}, + {"task_name": "judge", "task_handle": "judge-handle"}, + {"command": "echo no-names"}, # missing both keys -> contributes nothing + ] + + names = RayBackend._compute_in_batch_dep_names(jobs) + + assert names == {"train", "judge", "train-handle", "judge-handle"} + + +def test_deps_satisfied_false_for_in_batch_dep_not_yet_completed(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + # Premature-start guard: dep is in flight in this batch and not SUCCEEDED. + in_batch = {"train", "train-handle"} + assert RayBackend._deps_satisfied(["train"], {}, in_batch) is False + # Recorded under a non-success terminal state still blocks. + assert RayBackend._deps_satisfied(["train"], {"train": "FAILED"}, in_batch) is False + + +def test_deps_satisfied_true_when_in_batch_dep_succeeded_by_task_name_or_handle(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + in_batch = {"train", "train-handle"} + # Recorded SUCCEEDED under task_name. + assert RayBackend._deps_satisfied(["train"], {"train": "SUCCEEDED"}, in_batch) is True + # Recorded SUCCEEDED under the nemo-run handle. + assert RayBackend._deps_satisfied(["train-handle"], {"train-handle": "SUCCEEDED"}, in_batch) is True + + +def test_deps_satisfied_true_for_cross_experiment_dep_gated_upstream(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + # Dep matches no job in this batch -> gated upstream -> treated satisfied. + in_batch = {"judge", "judge-handle"} + assert RayBackend._deps_satisfied(["prior-experiment-handle"], {}, in_batch) is True + + +def test_deps_satisfied_mixed_in_batch_pending_and_cross_experiment(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + in_batch = {"train", "train-handle"} + deps = ["train", "prior-experiment-handle"] + # Blocked while the in-batch dep is still pending, even though the other is gated. + assert RayBackend._deps_satisfied(deps, {}, in_batch) is False + # Unblocks once the in-batch dep succeeds. + assert RayBackend._deps_satisfied(deps, {"train": "SUCCEEDED"}, in_batch) is True + + +def test_deps_satisfied_true_for_empty_or_none_deps(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + assert RayBackend._deps_satisfied([], {}, set()) is True + assert RayBackend._deps_satisfied(None, {}, set()) is True + + +# --------------------------------------------------------------------------- +# Poll-failure cleanup +# --------------------------------------------------------------------------- + + +def test_handle_poll_failure_stops_jobs_once_and_chains_exception(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + backend = RayBackend(dashboard_url="http://ray-head:8265") + + calls = [] + + def _record(client, job_ids, *, reason): + calls.append((client, list(job_ids), reason)) + + backend._stop_jobs_best_effort = _record # type: ignore[method-assign] + + client = object() + job_ids = ["job-1", "job-2"] + original = ValueError("poll boom") + + with pytest.raises(RuntimeError) as excinfo: + backend._handle_poll_failure(client, job_ids, original) + + # Cleanup attempted exactly once for the given job ids with the poll-failure reason. + assert calls == [(client, ["job-1", "job-2"], "poll-failure")] + # Wrapped error is chained from the original via ``from exc``. + assert excinfo.value.__cause__ is original + + +def test_stop_jobs_best_effort_continues_when_one_stop_raises(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + backend = RayBackend(dashboard_url="http://ray-head:8265") + + stopped = [] + + class FlakyClient: + def stop_job(self, job_id): + stopped.append(job_id) + if job_id == "job-1": + raise RuntimeError("stop failed") + + def get_job_status(self, job_id): + # Report terminal immediately so verification does not block. + return "STOPPED" + + # Should not propagate the per-job stop error and should attempt every job. + backend._stop_jobs_best_effort(FlakyClient(), ["job-1", "job-2"], reason="poll-failure") + + assert sorted(stopped) == ["job-1", "job-2"] From dc677647c3165509096dbccaaf28c232ff22da8f Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 9 Jun 2026 17:08:35 -0400 Subject: [PATCH 12/46] fix(backend): honor dashboard_url precreated clusters in metadata and the multi-node gate stage_metadata() and the non-Slurm multi-node guard treated a precreated cluster as reachable only via an endpoint, so a dashboard_url-only (Jobs API) config still requested an embedded Ray cluster and could fail the multi-node check. Use the resolved backend and accept dashboard_url alongside endpoint in both places. Add a regression test that a dashboard_url-only config resolves as precreated and does not request an embedded cluster. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/exp.py | 18 +++++++++--------- nemo_skills/pipeline/utils/ray_backend.py | 7 ++++++- tests/test_backends.py | 16 ++++++++++++++++ 3 files changed, 31 insertions(+), 10 deletions(-) diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index fab5e11c9b..b61d4539e0 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -242,7 +242,8 @@ def get_executor( Raised if a non-SLURM executor is requested with `num_nodes > 1`. """ env_vars = get_env_variables(cluster_config) - backend_env = get_execution_backend(cluster_config, with_ray=with_ray).get_env_overrides() + backend = get_execution_backend(cluster_config, with_ray=with_ray) + backend_env = backend.get_env_overrides() if backend_env: env_vars.update(backend_env) config_mounts = get_mounts_from_config(cluster_config) @@ -253,15 +254,14 @@ def get_executor( extra_package_dirs = tuple(extra_package_dirs) packager = get_packager(extra_package_dirs=extra_package_dirs) - # Ray backend with a precreated cluster allows multi-node without SLURM. - backend_config = cluster_config.get("backend") or cluster_config.get("execution_backend") or {} - if isinstance(backend_config, str): - backend_config = {"name": backend_config} - backend_name = str(backend_config.get("name", "")).strip().lower() - is_ray_precreated = backend_name == "ray" and bool(backend_config.get("precreated_cluster", False)) - if is_ray_precreated and not backend_config.get("endpoint"): + # Ray backend with a precreated cluster allows multi-node without SLURM. Use the + # resolved backend so precreated mode inferred from a dashboard_url, endpoint, or + # Kubernetes alias is honored, not just an explicit precreated_cluster flag. + is_ray_precreated = getattr(backend, "name", "") == "ray" and bool(getattr(backend, "precreated_cluster", False)) + if is_ray_precreated and not (getattr(backend, "endpoint", None) or getattr(backend, "dashboard_url", None)): raise ValueError( - "Invalid cluster_config: backend.precreated_cluster=true requires backend.endpoint to be set." + "Invalid cluster_config: backend.precreated_cluster=true requires " + "backend.endpoint or backend.dashboard_url to be set." ) if cluster_config["executor"] != "slurm": diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 1d3196c443..48c42326a6 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -381,7 +381,12 @@ def stage_metadata( use_with_ray_cluster: bool = False, container_image: str | None = None, ) -> Dict[str, Any] | None: - if self.precreated_cluster and self.endpoint: + """Build per-stage Ray metadata (cluster mode, address, placement labels). + + A precreated cluster reachable by an endpoint (Ray Client) or a + dashboard_url (Jobs API) must not request an embedded Ray cluster. + """ + if self.precreated_cluster and (self.endpoint or self.dashboard_url): should_use_embedded_ray_cluster = False else: should_use_embedded_ray_cluster = True diff --git a/tests/test_backends.py b/tests/test_backends.py index 85e76ee368..1d12ec1346 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -162,6 +162,22 @@ def test_ray_backend_forwards_required_env_vars_to_runtime_env(): assert runtime_env["env_vars"]["MY_JUDGE_KEY"] == "secret-123" +def test_dashboard_only_ray_config_is_precreated_not_embedded(): + # backend.name: ray + dashboard_url (no endpoint) targets a pre-provisioned + # cluster via the Jobs API, so it must resolve as precreated and must not + # request an embedded Ray cluster. + cluster_config = { + "executor": "slurm", + "backend": {"name": "ray", "dashboard_url": "http://ray-head:8265"}, + } + + backend = get_execution_backend(cluster_config) + assert backend.precreated_cluster is True + + metadata = backend.stage_metadata(container_image="nvcr.io/nvidia/pytorch:25.02-py3") + assert not (metadata or {}).get("use_with_ray_cluster") + + def test_ray_backend_runtime_env_normalizes_and_filters_values(): from nemo_skills.pipeline.utils.ray_backend import RayBackend From d3b259326b8024f257ceb78a74faff708cc949e2 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 9 Jun 2026 17:18:48 -0400 Subject: [PATCH 13/46] fix(ray-backend): honor use_with_ray_cluster for non-precreated stage metadata The non-precreated branch of stage_metadata hardcoded the embedded-cluster flag on, ignoring the caller's use_with_ray_cluster parameter. Initialize it from the parameter and only force it off for a precreated cluster reachable by endpoint or dashboard_url. Behavior is unchanged for the Ray callers (which pass True). Add a test that the flag is honored. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/ray_backend.py | 2 +- tests/test_backends.py | 11 +++++++++++ 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 48c42326a6..2d1b1a8a91 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -389,7 +389,7 @@ def stage_metadata( if self.precreated_cluster and (self.endpoint or self.dashboard_url): should_use_embedded_ray_cluster = False else: - should_use_embedded_ray_cluster = True + should_use_embedded_ray_cluster = use_with_ray_cluster metadata = super().stage_metadata(use_with_ray_cluster=should_use_embedded_ray_cluster) metadata = dict(metadata or {}) diff --git a/tests/test_backends.py b/tests/test_backends.py index 1d12ec1346..26a4353854 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -178,6 +178,17 @@ def test_dashboard_only_ray_config_is_precreated_not_embedded(): assert not (metadata or {}).get("use_with_ray_cluster") +def test_non_precreated_ray_honors_use_with_ray_cluster_flag(): + # Without a precreated cluster (no endpoint/dashboard_url), stage_metadata must + # honor the caller's use_with_ray_cluster flag rather than force it on. + cluster_config = {"executor": "slurm", "backend": {"name": "ray"}} + backend = get_execution_backend(cluster_config) + assert backend.precreated_cluster is False + + assert backend.stage_metadata(use_with_ray_cluster=True).get("use_with_ray_cluster") is True + assert not (backend.stage_metadata(use_with_ray_cluster=False) or {}).get("use_with_ray_cluster") + + def test_ray_backend_runtime_env_normalizes_and_filters_values(): from nemo_skills.pipeline.utils.ray_backend import RayBackend From 9026523749c030b90602e5a5d91b77e40a9e6c88 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 9 Jun 2026 17:33:37 -0400 Subject: [PATCH 14/46] docs(backend): add docstrings to Ray backend and execution-backend APIs Docstrings only; no behavior, signature, or import changes. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/backends.py | 11 +++++++++++ nemo_skills/pipeline/utils/exp.py | 2 ++ nemo_skills/pipeline/utils/ray_backend.py | 17 +++++++++++++++++ 3 files changed, 30 insertions(+) diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index d908ebacc9..39993ea49c 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -80,14 +80,17 @@ def stage_metadata( use_with_ray_cluster: bool = False, container_image: str | None = None, ) -> Dict[str, Any] | None: + """Return per-stage script metadata, or None when no metadata applies.""" if use_with_ray_cluster: return {"use_with_ray_cluster": True} return None def get_env_overrides(self) -> Dict[str, str]: + """Return environment variable overrides to inject into stage commands.""" return {} def start_experiment(self, exp: run.Experiment, cluster_config: Dict[str, Any], options: BackendRunOptions): + """Run the experiment, detaching for Slurm and tailing logs otherwise.""" if options.dry_run: LOG.info("Dry run mode is enabled, not running the experiment.") return @@ -98,6 +101,7 @@ def start_experiment(self, exp: run.Experiment, cluster_config: Dict[str, Any], exp.run(detach=True, sequential=options.sequential) def track_experiment(self, exp: run.Experiment, include_finished: bool = True) -> Dict[str, Any]: + """Return a task-name to status map, optionally filtering to active tasks.""" status_dict = exp.status(return_dict=True) if include_finished: return status_dict @@ -115,6 +119,7 @@ def track_experiment(self, exp: run.Experiment, include_finished: bool = True) - } def stop_experiment(self, exp: run.Experiment, only_active: bool = True) -> list[str]: + """Cancel experiment tasks and return the names of the cancelled tasks.""" cancelled_jobs = [] active_states = { "RUNNING", @@ -275,16 +280,22 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa def _with_exp(exp_or_name: run.Experiment | str): + """Return a context manager yielding an Experiment from an instance, title, or id.""" if isinstance(exp_or_name, run.Experiment): class _ExperimentCtx: + """Context manager that yields an already-instantiated experiment unchanged.""" + def __init__(self, exp): + """Store the experiment to yield from the context.""" self.exp = exp def __enter__(self): + """Return the wrapped experiment.""" return self.exp def __exit__(self, exc_type, exc, tb): + """Leave the context without suppressing exceptions.""" return False return _ExperimentCtx(exp_or_name) diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index b61d4539e0..0ff73fd476 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -697,6 +697,7 @@ def add_task( executors = [] def add_server_tasks(): + """Append the server (and its executor) for each requested server replica.""" nonlocal het_group # avoid mutating server_config, as it may be used again later in dependent jobs _server_config = copy.deepcopy(server_config) @@ -992,6 +993,7 @@ def queue_ray_job_commands( before_count = len(queued_jobs) def _dep_name(dep): + """Return a dependency's name, accepting either a string or a handle object.""" return dep if isinstance(dep, str) else getattr(dep, "name", str(dep)) all_deps = list(task_dependencies or []) + list(external_dependencies or []) diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 2d1b1a8a91..99b3e509f5 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -93,6 +93,7 @@ def __init__( image_label_selectors: Dict[str, Dict[str, str]] | None = None, env_vars: Dict[str, str] | None = None, ): + """Configure connection, placement, and env-forwarding options for the backend.""" self.endpoint = endpoint.strip() if endpoint else None self.dashboard_url = self._normalize_dashboard_url(dashboard_url, self.endpoint) self.precreated_cluster = precreated_cluster @@ -123,6 +124,7 @@ def __init__( @staticmethod def _normalize_dashboard_url(dashboard_url: str | None, endpoint: str | None) -> str | None: + """Return the Jobs API dashboard URL, deriving it from the endpoint if unset.""" if dashboard_url: return str(dashboard_url).strip() if not endpoint: @@ -138,11 +140,13 @@ def _normalize_dashboard_url(dashboard_url: str | None, endpoint: str | None) -> @staticmethod def _sanitize_submission_id(value: str) -> str: + """Return a submission id with unsupported characters replaced by hyphens.""" sanitized = re.sub(r"[^a-zA-Z0-9_.-]", "-", value) return sanitized.strip("-._") or "job" @staticmethod def _to_status_str(status: Any) -> str: + """Normalize a Ray job status (enum or string) to an uppercase string.""" return str(getattr(status, "value", status)).upper() @staticmethod @@ -184,6 +188,7 @@ def _deps_satisfied( return True def _get_jobs_client(self): + """Return a JobSubmissionClient for the dashboard URL, raising if unavailable.""" if not self.dashboard_url: raise RuntimeError("Ray Jobs submission requires backend.dashboard_url (e.g. http://:8265).") try: @@ -331,6 +336,7 @@ def _append_stage_job_record( @staticmethod def _prepare_job_entrypoint_command(command: str) -> str: + """Rewrite a command to use local cluster autodiscovery (RAY_ADDRESS=auto).""" # When running *inside* a Ray Job, the driver is already on the cluster. # Replace any forwarded Ray Client endpoint with local cluster autodiscovery. stripped = re.sub(r"export\s+RAY_ADDRESS=[^&;]+&&\s*", "", command) @@ -340,6 +346,7 @@ def _prepare_job_entrypoint_command(command: str) -> str: def _normalize_image_label_selectors( selectors: Dict[str, Dict[str, str]] | None, ) -> Dict[str, Dict[str, str]]: + """Validate and normalize image-pattern label selectors into a uniform dict.""" if selectors is None: return {} if not isinstance(selectors, dict): @@ -358,6 +365,7 @@ def _normalize_image_label_selectors( return normalized def _labels_for_image(self, container_image: str) -> Dict[str, str]: + """Return placement labels for a container image by glob-matching selectors.""" labels: Dict[str, str] = {} for pattern, selector in self.image_label_selectors.items(): if fnmatch.fnmatch(container_image, pattern): @@ -367,6 +375,7 @@ def _labels_for_image(self, container_image: str) -> Dict[str, str]: @staticmethod def _normalize_label_value(value: str) -> str: + """Return a label value lowercased with unsupported characters collapsed to hyphens.""" normalized = re.sub(r"[^a-z0-9_.-]", "-", value.lower()) normalized = re.sub(r"-+", "-", normalized).strip("-._") return normalized or "unknown" @@ -419,6 +428,7 @@ def stage_metadata( return metadata def get_env_overrides(self) -> Dict[str, str]: + """Return RAY_ADDRESS pointing at the endpoint, or nothing when unset.""" if not self.endpoint: return {} return {"RAY_ADDRESS": self.endpoint} @@ -429,6 +439,7 @@ def get_env_overrides(self) -> Dict[str, str]: @staticmethod def _preflight_config(cluster_config: Dict[str, Any]) -> Dict[str, Any]: + """Return the validated ``preflight`` section of the cluster config.""" cfg = cluster_config.get("preflight") or {} if cfg is None: return {} @@ -438,6 +449,7 @@ def _preflight_config(cluster_config: Dict[str, Any]) -> Dict[str, Any]: @staticmethod def _normalize_resource_key(key: str) -> str: + """Map common resource aliases (gpu/cpu/mem) to canonical Ray resource keys.""" k = str(key).strip().lower() if k in {"gpu", "gpus"}: return "GPU" @@ -449,6 +461,7 @@ def _normalize_resource_key(key: str) -> str: @staticmethod def _extract_node_labels(nodes_detail: list[Any]) -> list[Dict[str, str]]: + """Extract per-node label maps from Ray state API node-detail entries.""" label_maps: list[Dict[str, str]] = [] for node in nodes_detail: if hasattr(node, "model_dump"): @@ -465,6 +478,7 @@ def _extract_node_labels(nodes_detail: list[Any]) -> list[Dict[str, str]]: return label_maps def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOptions) -> None: + """Verify cluster reachability, resources, and node labels before submitting.""" preflight_cfg = self._preflight_config(cluster_config) enabled = bool(preflight_cfg.get("enabled", True)) if not enabled: @@ -579,6 +593,7 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti # ------------------------------------------------------------------ def start_experiment(self, exp: run.Experiment, cluster_config: Dict[str, Any], options: BackendRunOptions): + """Run preflight, then submit queued Ray Jobs concurrently and wait for them.""" self._run_preflight(cluster_config, options) pending_jobs = getattr(exp, "_ns_ray_jobs_queue", None) if not pending_jobs: @@ -814,6 +829,7 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: setattr(exp, "_ns_ray_jobs_submitted", submitted_meta) def track_experiment(self, exp: run.Experiment, include_finished: bool = True) -> Dict[str, Any]: + """Return current status for each submitted Ray Job, optionally active-only.""" submitted = getattr(exp, "_ns_ray_jobs_submitted", None) if not submitted or not self.dashboard_url: return super().track_experiment(exp, include_finished=include_finished) @@ -840,6 +856,7 @@ def track_experiment(self, exp: run.Experiment, include_finished: bool = True) - return tracked def stop_experiment(self, exp: run.Experiment, only_active: bool = True) -> list[str]: + """Stop submitted Ray Jobs and return the names of the stopped tasks.""" submitted = getattr(exp, "_ns_ray_jobs_submitted", None) if not submitted or not self.dashboard_url: return super().stop_experiment(exp, only_active=only_active) From 169ce824b3ef539d46421345aa9e0ae9a47bb4e7 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Wed, 10 Jun 2026 10:13:34 -0400 Subject: [PATCH 15/46] fix(backend): accept dashboard_url for precreated Ray preflight A dashboard-only Jobs API cluster (name=ray + dashboard_url, no endpoint) is a valid precreated cluster, but preflight required an endpoint whenever precreated_cluster was set. Accept endpoint OR dashboard_url (matching stage_metadata and the multi-node gate); reachability checks default to whether an endpoint exists and are skipped when only dashboard_url is configured. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/ray_backend.py | 21 +++++++++++++---- tests/test_backends.py | 28 +++++++++++++++++++++++ 2 files changed, 44 insertions(+), 5 deletions(-) diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 99b3e509f5..19ee9201d8 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -484,8 +484,11 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti if not enabled: return - if self.precreated_cluster and not self.endpoint: - raise RuntimeError("Ray preflight failed: backend.precreated_cluster=true requires backend.endpoint.") + if self.precreated_cluster and not (self.endpoint or self.dashboard_url): + raise RuntimeError( + "Ray preflight failed: backend.precreated_cluster=true requires " + "backend.endpoint or backend.dashboard_url." + ) if options.dry_run: return @@ -495,14 +498,22 @@ def _run_preflight(self, cluster_config: Dict[str, Any], options: BackendRunOpti required_labels = preflight_cfg.get("required_node_labels") or [] min_resources = preflight_cfg.get("min_cluster_resources") or {} - require_reachable = bool(preflight_cfg.get("require_ray_endpoint", self.precreated_cluster)) + require_reachable = bool(preflight_cfg.get("require_ray_endpoint", bool(self.endpoint))) strict_label_check = bool(preflight_cfg.get("strict_label_check", True)) if not require_reachable and not required_labels and not min_resources: return - if require_reachable and not self.endpoint: - raise RuntimeError("Ray preflight failed: backend.endpoint is required for connectivity checks.") + # Reachability, resource, and label inspection all connect via ray.init(address=endpoint); + # a dashboard-only (Jobs API) precreated cluster has no Ray Client endpoint to inspect. + if not self.endpoint: + if require_reachable: + raise RuntimeError("Ray preflight failed: backend.endpoint is required for connectivity checks.") + LOG.warning( + "Ray preflight: skipping resource/label inspection because backend.endpoint is unset " + "(dashboard-only Jobs API cannot drive ray.nodes())." + ) + return try: import ray diff --git a/tests/test_backends.py b/tests/test_backends.py index 26a4353854..27f3accbe8 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -178,6 +178,34 @@ def test_dashboard_only_ray_config_is_precreated_not_embedded(): assert not (metadata or {}).get("use_with_ray_cluster") +def test_dashboard_only_precreated_preflight_does_not_require_endpoint(): + # A dashboard-only (Jobs API) precreated cluster has no Ray Client endpoint, + # so preflight must not require one: require_ray_endpoint now defaults to + # bool(endpoint)=False, so the checks short-circuit without touching ray.init. + from nemo_skills.pipeline.utils.backends import BackendRunOptions + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + backend = RayBackend(dashboard_url="http://ray-head:8265", precreated_cluster=True) + assert backend.endpoint is None + + # No preflight section -> defaults; dry_run=False to exercise the require_reachable default. + backend._run_preflight({}, BackendRunOptions(dry_run=False)) + + +def test_dashboard_only_explicit_require_endpoint_still_raises(): + # An explicit preflight.require_ray_endpoint=true is still honored: a connectivity + # check needs a Ray Client endpoint, which a dashboard-only cluster lacks. + from nemo_skills.pipeline.utils.backends import BackendRunOptions + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + backend = RayBackend(dashboard_url="http://ray-head:8265", precreated_cluster=True) + assert backend.endpoint is None + + cluster_config = {"preflight": {"require_ray_endpoint": True}} + with pytest.raises(RuntimeError, match="endpoint"): + backend._run_preflight(cluster_config, BackendRunOptions(dry_run=False)) + + def test_non_precreated_ray_honors_use_with_ray_cluster_flag(): # Without a precreated cluster (no endpoint/dashboard_url), stage_metadata must # honor the caller's use_with_ray_cluster flag rather than force it on. From a7a61efa94069605262e5b8a32056ac6a8853f91 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Wed, 10 Jun 2026 14:56:54 -0400 Subject: [PATCH 16/46] fix(declarative): filter finished cross-experiment deps out of exp.add A run_after naming another experiment whose tasks already finished resolves to empty handles and falls through as a bare experiment-name string in internal_deps (the _reuse_exp path). nemo-run's exp.add asserts every dependency is a job in the current experiment, raising "Dependency not found" on the Mode-3 declarative eval path. Filter internal_deps against the experiment's own job ids (only when exp.jobs is a concrete list) before exp.add; cross-experiment ordering is still honored via external deps and the Ray queue resolver. Adds a regression test that simulates exp.jobs growth -- the prior dependency test mocked get_exp_handles to always return non-empty, hiding this path. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/declarative.py | 24 ++++++++ tests/test_declarative_pipeline.py | 75 +++++++++++++++++++++++ 2 files changed, 99 insertions(+) diff --git a/nemo_skills/pipeline/utils/declarative.py b/nemo_skills/pipeline/utils/declarative.py index 4ce41b5bbb..6386d1fe53 100644 --- a/nemo_skills/pipeline/utils/declarative.py +++ b/nemo_skills/pipeline/utils/declarative.py @@ -938,6 +938,30 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: should_use_with_ray_cluster=should_use_with_ray_cluster, ) + # A run_after naming another experiment whose tasks have already finished + # resolves to empty handles and falls through as a bare experiment-name string + # in internal_deps. nemo-run's exp.add asserts every dependency is a job in THIS + # experiment, so drop deps that are not present here -- cross-experiment ordering + # is still honored via external_deps and (on Ray) the queued dep resolver. Only + # filters when exp.jobs is introspectable, so it never over-drops valid handles. + # Only filter when exp.jobs is a concrete list (a real nemo-run experiment); a + # mocked or duck-typed exp leaves deps untouched so valid handles are never dropped. + exp_jobs = getattr(exp, "jobs", None) + if internal_deps and isinstance(exp_jobs, (list, tuple)): + known_job_ids = {getattr(job, "id", None) for job in exp_jobs} + kept = [] + for dep in internal_deps: + if not isinstance(dep, str) or dep in known_job_ids: + kept.append(dep) + else: + LOG.warning( + "Dropping dependency '%s' from exp.add: not a job in this " + "experiment (cross-experiment ordering preserved via external " + "deps / Ray queue).", + dep, + ) + internal_deps = kept or None + # Add to experiment and return task ID # Note: Internal dependencies (task handles from same experiment) go to exp.add() # External dependencies (SLURM job IDs from other experiments) go to executor diff --git a/tests/test_declarative_pipeline.py b/tests/test_declarative_pipeline.py index 3868412f0b..6cae8f682f 100644 --- a/tests/test_declarative_pipeline.py +++ b/tests/test_declarative_pipeline.py @@ -722,6 +722,81 @@ def mock_get_executor(**kwargs): call2_kwargs = mock_exp.add.call_args_list[1][1] assert call2_kwargs["dependencies"] == ["task_handle_1"] + def test_finished_cross_experiment_dep_filtered_from_exp_add(self): + """A run_after naming another experiment whose tasks already finished resolves to + empty handles and falls through as a bare experiment-name string in internal_deps + (the _reuse_exp path). It must be dropped from exp.add (nemo-run asserts every dep + is a job in THIS experiment) while same-experiment handles are kept. Reproduces the + bug the legacy cross-experiment dependency patch fixed, now handled natively. + """ + import nemo_run as run + + # Empty handles -> the upstream experiment already finished / does not exist. + with patch("nemo_skills.pipeline.utils.declarative.get_exp_handles") as mock_get_handles: + mock_get_handles.return_value = [] + + with patch("nemo_skills.pipeline.utils.declarative.get_exp") as mock_get_exp: + mock_exp = MagicMock(spec=run.Experiment) + mock_exp.__enter__ = MagicMock(return_value=mock_exp) + mock_exp.__exit__ = MagicMock(return_value=False) + # Simulate nemo-run: each add() appends a Job (id == handle) to exp.jobs. + mock_exp.jobs = [] + + def fake_add(*args, **kwargs): + handle = f"task_handle_{len(mock_exp.jobs) + 1}" + job = MagicMock() + job.id = handle + mock_exp.jobs.append(job) + return handle + + mock_exp.add = MagicMock(side_effect=fake_add) + mock_get_exp.return_value = mock_exp + + def mock_get_executor(**kwargs): + mock_executor = MagicMock() + mock_executor.packager = MagicMock() + return mock_executor + + with patch("nemo_skills.pipeline.utils.declarative.get_executor", side_effect=mock_get_executor): + with patch("nemo_skills.pipeline.utils.declarative.run_exp"): + cluster_config = { + "executor": "slurm", + "containers": {"nemo-skills": "test/container"}, + "account": "test", + "env_vars": {"HF_HOME": "/mounted/hf_home"}, + "mounts": ["/mounted/hf_home:/mounted/hf_home"], + } + + cmd1 = make_command(inline="echo job1", name="job1") + group1 = CommandGroup(commands=[cmd1], name="group1", log_dir="/tmp/logs") + cmd2 = make_command(inline="echo job2", name="job2") + group2 = CommandGroup(commands=[cmd2], name="group2", log_dir="/tmp/logs") + + job1_spec = {"name": "job1", "group": group1} + # job2 depends on job1 (internal) AND a finished external experiment (string). + job2_spec = { + "name": "job2", + "group": group2, + "dependencies": [job1_spec, "finished_external_experiment"], + } + + pipeline = Pipeline( + name="test_pipeline", + cluster_config=cluster_config, + jobs=[job1_spec, job2_spec], + skip_hf_home_check=True, + reuse_code=False, + ) + + # _reuse_exp set -> empty-handle string deps take the internal path. + pipeline.run(dry_run=True, _reuse_exp=mock_exp) + + assert mock_exp.add.call_count == 2 + # job1: no dependencies. + assert mock_exp.add.call_args_list[0][1]["dependencies"] is None + # job2: same-experiment handle kept, finished cross-exp string dropped. + assert mock_exp.add.call_args_list[1][1]["dependencies"] == ["task_handle_1"] + def test_run_after_dependencies_across_experiments(self, tmp_path): """Test that run_after dependencies work when chaining multiple generate/run_cmd calls. From f8697ec3c057f949988a634ecb12928a20ce1d63 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Wed, 10 Jun 2026 14:56:54 -0400 Subject: [PATCH 17/46] chore(backends): remove dead _RayBackendShim re-export shim _RayBackendShim was defined then immediately deleted and never referenced; RayBackend is already re-exported via the import and __all__. Remove the dead class and the misleading placeholder comment. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/backends.py | 11 +---------- 1 file changed, 1 insertion(+), 10 deletions(-) diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index 39993ea49c..d7aa061f16 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -146,16 +146,7 @@ def stop_experiment(self, exp: run.Experiment, only_active: bool = True) -> list # --------------------------------------------------------------------------- from nemo_skills.pipeline.utils.ray_backend import RayBackend # noqa: E402 # re-exported - -class _RayBackendShim(RayBackend): - """Shim so that isinstance checks against the old import path still work.""" - - -# Keep the name RayBackend pointing at the canonical class. -del _RayBackendShim # only the alias is needed - - -# Placeholder so linters don't complain about the import being "unused". +# RayBackend is re-exported here (listed in __all__) for backwards-compatible imports. __all__ = [ "BackendRunOptions", "ExecutionBackend", From 0057fc25a501db969eb7199c6a4dd5a8c2369726 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Wed, 10 Jun 2026 17:33:31 -0400 Subject: [PATCH 18/46] chore(ci): re-trigger CI Signed-off-by: Nick Gupta From e40ace144a4a2ff44511737ded5e103b2a502ea8 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Thu, 11 Jun 2026 19:31:32 -0400 Subject: [PATCH 19/46] fix(ray): merge job + driver runtime_env via RAY_OVERRIDE_JOB_RUNTIME_ENV A Ray job whose driver re-inits Ray with its own runtime_env (NeMo-RL GRPO/rollout via run_grpo_nemo_gym.py -> init_ray) fails with "Failed to merge the Job's runtime env ... because of a conflict" when a key (OPENAI_API_KEY, HF_HOME, ...) is present in both the job-level runtime_env we forward and the driver's ray.init runtime_env. Always set RAY_OVERRIDE_JOB_RUNTIME_ENV=1 in the submitted runtime_env so Ray merges (driver wins) instead of erroring. No effect on jobs that never re-init Ray. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/ray_backend.py | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 19ee9201d8..0acc439ffe 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -200,8 +200,18 @@ def _get_jobs_client(self): return JobSubmissionClient(self.dashboard_url) def _build_runtime_env(self) -> Dict[str, Any] | None: - """runtime_env that forwards cluster env vars (API keys, HF_TOKEN, ...) to a job.""" - return {"env_vars": dict(self.env_vars)} if self.env_vars else None + """runtime_env that forwards cluster env vars (API keys, HF_TOKEN, ...) to a job. + + Always sets RAY_OVERRIDE_JOB_RUNTIME_ENV=1. A job whose driver calls ray.init() + again with its own runtime_env (e.g. NeMo-RL GRPO/rollout) would otherwise hit + "Failed to merge the Job's runtime env ... because of a conflict" whenever a key + (OPENAI_API_KEY, HF_HOME, ...) appears in both this job-level runtime_env and the + driver's ray.init() runtime_env. The flag tells Ray to merge (driver wins) instead + of erroring. No effect on jobs that never re-init Ray. + """ + env_vars = dict(self.env_vars) + env_vars["RAY_OVERRIDE_JOB_RUNTIME_ENV"] = "1" + return {"env_vars": env_vars} def _stop_jobs_best_effort( self, From 98f8f475e54e3208569fec28b3a377816e980e15 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Thu, 11 Jun 2026 20:01:44 -0400 Subject: [PATCH 20/46] test(ray): update _build_runtime_env tests for RAY_OVERRIDE_JOB_RUNTIME_ENV The runtime_env now always carries RAY_OVERRIDE_JOB_RUNTIME_ENV=1, so the normalize/filter test expects it in env_vars and the no-env case asserts the override dict instead of None. Signed-off-by: Nick Gupta --- tests/test_backends.py | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/tests/test_backends.py b/tests/test_backends.py index 27f3accbe8..3d874f5a10 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -222,15 +222,18 @@ def test_ray_backend_runtime_env_normalizes_and_filters_values(): backend = RayBackend(dashboard_url="http://ray-head:8265", env_vars={"A": "1", "B": None, "C": 2}) - assert backend._build_runtime_env() == {"env_vars": {"A": "1", "C": "2"}} + # RAY_OVERRIDE_JOB_RUNTIME_ENV is always injected so a job driver that re-inits Ray + # (e.g. NeMo-RL GRPO/rollout) can merge its runtime_env instead of erroring on a shared key. + assert backend._build_runtime_env() == {"env_vars": {"A": "1", "C": "2", "RAY_OVERRIDE_JOB_RUNTIME_ENV": "1"}} -def test_ray_backend_runtime_env_none_when_no_env(): +def test_ray_backend_runtime_env_sets_override_even_with_no_env(): from nemo_skills.pipeline.utils.ray_backend import RayBackend backend = RayBackend(dashboard_url="http://ray-head:8265", env_vars={}) - assert backend._build_runtime_env() is None + # Even with no forwarded env vars the runtime_env still carries the override flag (never None). + assert backend._build_runtime_env() == {"env_vars": {"RAY_OVERRIDE_JOB_RUNTIME_ENV": "1"}} # --------------------------------------------------------------------------- From be5a64fbd813a8d6a3051d29b7c4ef6ec86a4763 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Fri, 12 Jun 2026 14:57:50 -0400 Subject: [PATCH 21/46] feat(ray): fail fast when backend.name=ray + executor=none lacks a dashboard_url A 'backend.name: ray' config with 'executor: none' targets a pre-provisioned Ray Jobs cluster and needs a dashboard URL to submit to. When none resolved, get_execution_backend silently treated it as a non-precreated (Ray-inside-Slurm) backend, which then failed deep in nemo-run with the opaque 'use_with_ray_cluster is only supported for SlurmExecutor'. Raise an actionable error up front pointing at backend.dashboard_url. Scoped to executor: none so the valid Ray-inside-Slurm path (executor: slurm, no dashboard) is unaffected. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/backends.py | 14 ++++++++++++++ tests/test_backends.py | 9 +++++++++ 2 files changed, 23 insertions(+) diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index d7aa061f16..b3178e7036 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -200,6 +200,20 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa bool(backend_config.get("precreated_cluster", False)) or bool(endpoint) or bool(dashboard_url) ) control_plane = str(backend_config.get("control_plane") or "").strip().lower() + # `backend.name: ray` with `executor: none` targets a pre-provisioned Ray Jobs + # cluster, which needs a dashboard URL to submit to. If none resolved (and this + # is not a Kubernetes control plane), fail fast with an actionable message rather + # than fall through to the opaque nemo-run error + # "use_with_ray_cluster is only supported for SlurmExecutor". + if not precreated_cluster and control_plane != "kubernetes": + executor_kind = str(cluster_config.get("executor") or "").strip().lower() + if executor_kind == "none": + raise ValueError( + "backend.name: ray with executor: none targets a pre-provisioned Ray Jobs " + "cluster and requires a dashboard URL — set " + "backend.dashboard_url: http://: (or backend.precreated_cluster: true). " + "To run Ray inside a Slurm allocation instead, set executor: slurm." + ) k8s_cfg = backend_config.get("kubernetes") or {} selector = backend_config.get("entrypoint_label_selector") or k8s_cfg.get("entrypoint_label_selector") image_label_key = backend_config.get("image_label_key") or k8s_cfg.get("image_label_key") diff --git a/tests/test_backends.py b/tests/test_backends.py index 3d874f5a10..94563ce944 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -217,6 +217,15 @@ def test_non_precreated_ray_honors_use_with_ray_cluster_flag(): assert not (backend.stage_metadata(use_with_ray_cluster=False) or {}).get("use_with_ray_cluster") +def test_ray_executor_none_without_dashboard_raises(): + # backend.name: ray + executor: none targets a pre-provisioned Ray Jobs cluster, + # which needs a dashboard URL. Without one (and no precreated flag), fail fast with + # an actionable message instead of the opaque nemo-run SlurmExecutor error. + cluster_config = {"executor": "none", "backend": {"name": "ray"}} + with pytest.raises(ValueError, match="dashboard_url"): + get_execution_backend(cluster_config) + + def test_ray_backend_runtime_env_normalizes_and_filters_values(): from nemo_skills.pipeline.utils.ray_backend import RayBackend From a351194d48c218b056774cdb1a995d03bdcdc72a Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Mon, 15 Jun 2026 17:45:35 -0400 Subject: [PATCH 22/46] fix(ray): use relative code paths in declarative rewrite under Ray backend For executor='none', _rewrite_local_paths rewrote /nemo_run/code to the driver's absolute nemo_skills install dir (e.g. /usr/local/lib/python3.10/dist-packages). That is correct for true run-on-this-host execution, but under the Ray Jobs API backend the command is submitted to a remote cluster whose image and Python version need not match the driver's, so that absolute path does not exist there -- the eval-generation job fails with "cd: /usr/local/lib/python3.10/dist-packages: No such file or directory". When backend.name == "ray", rewrite to relative (cwd) paths instead, mirroring the add_task path (which rewrites /nemo_run/code -> "./"). The job then runs from its working directory and nemo_skills resolves via the baked image's PYTHONPATH. Non-Ray behavior is unchanged. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/declarative.py | 22 ++++++++++++++++++---- 1 file changed, 18 insertions(+), 4 deletions(-) diff --git a/nemo_skills/pipeline/utils/declarative.py b/nemo_skills/pipeline/utils/declarative.py index 6386d1fe53..9f71373c9b 100644 --- a/nemo_skills/pipeline/utils/declarative.py +++ b/nemo_skills/pipeline/utils/declarative.py @@ -581,13 +581,27 @@ def _prepare_command(self, command, cluster_config: Dict) -> Tuple[run.Script, D return script, exec_config def _rewrite_local_paths(self, script: run.Script) -> run.Script: - """For executor='none', replace /nemo_run/code paths with local repo paths.""" + """For executor='none', replace /nemo_run/code paths with local repo paths. + + Under the Ray Jobs API backend (``backend.name == "ray"``) the command is + submitted to a *remote* cluster whose image and Python version need not match + the driver's, so the driver-side absolute install path (e.g. + ``/usr/local/lib/python3.10/dist-packages``) will not exist there. In that + case rewrite to relative (cwd) paths instead -- mirroring the add_task path -- + so the command runs from the job's working directory and ``nemo_skills`` + resolves via the baked image's ``PYTHONPATH``. + """ nemo_repo = get_registered_external_repo("nemo_skills") - if nemo_repo is None: + is_ray = (self.cluster_config.get("backend") or {}).get("name") == "ray" + if nemo_repo is None and not is_ray: return script - pkg_path = str(nemo_repo.path) - repo_root = str(nemo_repo.path.parent) + if is_ray: + pkg_path = "./nemo_skills" + repo_root = "." + else: + pkg_path = str(nemo_repo.path) + repo_root = str(nemo_repo.path.parent) def _replace(cmd: str) -> str: return cmd.replace("/nemo_run/code/nemo_skills", pkg_path).replace("/nemo_run/code", repo_root) From 80d9c9156dd2166f67416faef9b311e4f476a0b4 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Wed, 24 Jun 2026 10:56:31 -0400 Subject: [PATCH 23/46] fix(ray-backend): only pass entrypoint_label_selector when set and supported Ray's JobSubmissionClient.submit_job() gained entrypoint_label_selector in Ray 2.55; older clients (e.g. Ray 2.54 in the nemo-rl image) raise TypeError when it is passed, even as None. Build the submit_job kwargs incrementally and include entrypoint_label_selector only when a selector is configured and the installed Ray client's signature supports it (otherwise log a warning and proceed). The common no-selector path now works against any Ray version the cluster runs, while label routing is preserved on Ray >= 2.55. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/ray_backend.py | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 0acc439ffe..1a8aeabb1f 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -37,6 +37,7 @@ from __future__ import annotations import fnmatch +import inspect import json import logging import os @@ -731,13 +732,26 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: self.dashboard_url, selector or {}, ) - job_id = client.submit_job( + submit_kwargs = dict( entrypoint=entrypoint, submission_id=submission_id, metadata=metadata, - entrypoint_label_selector=selector or None, runtime_env=self._build_runtime_env(), ) + # entrypoint_label_selector was added to JobSubmissionClient.submit_job in + # Ray 2.55; older clients (e.g. Ray 2.54 in the nemo-rl image) reject the + # kwarg. Only pass it when a selector is configured and the client supports + # it, so the common no-selector path works against any Ray the cluster runs. + if selector: + if "entrypoint_label_selector" in inspect.signature(client.submit_job).parameters: + submit_kwargs["entrypoint_label_selector"] = selector + else: + LOG.warning( + "Ray client lacks entrypoint_label_selector support; ignoring " + "selector %s (requires Ray >= 2.55).", + selector, + ) + job_id = client.submit_job(**submit_kwargs) return { "task_name": task_name, "task_handle": job.get("task_handle"), From fb18e029ba01877c81bf0253686d910a472a9fd9 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Fri, 26 Jun 2026 10:44:58 -0400 Subject: [PATCH 24/46] fix(ray): keep the Ray Jobs backend off the non-Ray code path (review) Addresses review feedback that a few Ray-motivated refactors changed behavior for users who never opt into Ray. Each fix is gated so the default / Slurm-without-Ray path matches origin/main. - run_exp: honor --dry_run before the live mount check (it opens an SSH tunnel and stats sources), so a dry run stays offline-safe on every non-Ray caller. - declarative: scope the finished-cross-experiment internal_deps drop to the Ray backend; the default backend keeps nemo-run's fail-loud "Dependency not found" instead of silently dropping a dependency. - cli: drop the unrelated nemo_evaluator ModuleNotFoundError guard (reverts to the plain import; out of scope for this PR). - add_task / declarative: gate use_with_ray_cluster on executor == "slurm" so legacy with_ray on a non-Slurm executor no longer requests an embedded Ray cluster (nemo-run asserts SlurmExecutor). Precreated Jobs-API clusters never embed -- RayBackend.stage_metadata already enforces that. - _rewrite_local_paths / generate: parse the backend name through a shared get_backend_name() helper so the supported bare-string form (backend: ray) no longer raises AttributeError, and the duplicated parsing is removed. - add_task: only resolve command_images when a Ray backend is active (the list is consumed solely by the Ray queue); avoids a docker inspect per command on the non-Ray path for dockerfile: specs. - backends: import RayBackend lazily (only when a Ray backend is selected) via a local import + PEP 562 __getattr__, so the default path no longer imports the ~900-line ray_backend module. TYPE_CHECKING keeps the __all__ re-export valid. Tests: add regression coverage for the dry-run guard, the ray-vs-default cross-experiment dep handling, with_ray+non-Slurm metadata, Ray-only command-image resolution, the backend-name helpers, and the lazy import; reframe the existing cross-exp dep test as the Ray-backend case. Signed-off-by: Nick Gupta --- nemo_skills/pipeline/cli.py | 8 +- nemo_skills/pipeline/generate.py | 6 +- nemo_skills/pipeline/utils/backends.py | 52 ++++++++++++- nemo_skills/pipeline/utils/declarative.py | 25 +++++-- nemo_skills/pipeline/utils/exp.py | 32 ++++++-- tests/test_backends.py | 76 ++++++++++++++++++- tests/test_declarative_pipeline.py | 91 +++++++++++++++++++++-- tests/test_pipeline_utils.py | 87 ++++++++++++++++++++++ 8 files changed, 341 insertions(+), 36 deletions(-) diff --git a/nemo_skills/pipeline/cli.py b/nemo_skills/pipeline/cli.py index e83dffb2a7..d814dc8945 100644 --- a/nemo_skills/pipeline/cli.py +++ b/nemo_skills/pipeline/cli.py @@ -25,6 +25,7 @@ from nemo_skills.pipeline.eval import eval from nemo_skills.pipeline.generate import generate from nemo_skills.pipeline.megatron_lm.train import train_megatron_lm +from nemo_skills.pipeline.nemo_evaluator import nemo_evaluator from nemo_skills.pipeline.nemo_gym_rollouts import nemo_gym_rollouts from nemo_skills.pipeline.nemo_rl.grpo import grpo_nemo_rl from nemo_skills.pipeline.nemo_rl.sft import sft_nemo_rl @@ -37,13 +38,6 @@ from nemo_skills.pipeline.summarize_robustness import summarize_robustness from nemo_skills.pipeline.verl.ppo import ppo_verl -try: - from nemo_skills.pipeline.nemo_evaluator import nemo_evaluator -except ModuleNotFoundError as e: - # Keep non-evaluator commands usable when optional launcher deps are absent. - if e.name != "nemo_evaluator_launcher": - raise - typer.main.get_command_name = lambda name: name diff --git a/nemo_skills/pipeline/generate.py b/nemo_skills/pipeline/generate.py index 69f3110618..e787b749b8 100644 --- a/nemo_skills/pipeline/generate.py +++ b/nemo_skills/pipeline/generate.py @@ -22,6 +22,7 @@ from nemo_skills.dataset.utils import import_from_path from nemo_skills.inference import GENERATION_MODULE_MAP, GenerationType from nemo_skills.pipeline.app import app, typer_unpacker +from nemo_skills.pipeline.utils.backends import get_backend_name from nemo_skills.pipeline.utils.cluster import parse_kwargs from nemo_skills.pipeline.utils.declarative import ( Command, @@ -649,10 +650,7 @@ def convert_server_type_to_string(server_type): if not jobs: return None - backend_config = cluster_config.get("backend") or cluster_config.get("execution_backend") or {} - if isinstance(backend_config, str): - backend_config = {"name": backend_config} - with_ray_pipeline = str(backend_config.get("name", "")).strip().lower() == "ray" + with_ray_pipeline = get_backend_name(cluster_config) == "ray" # Create and run pipeline pipeline = Pipeline( diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index b3178e7036..851531ed8e 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -16,13 +16,19 @@ import logging from dataclasses import dataclass -from typing import Any, Dict +from typing import TYPE_CHECKING, Any, Dict import nemo_run as run from nemo_skills.pipeline.utils.cluster import get_env_variables from nemo_skills.utils import get_logger_name +if TYPE_CHECKING: + # Re-exported via __getattr__ at runtime (kept lazy); declared here so type checkers + # and ruff's __all__ resolution see RayBackend as a defined name without importing + # the heavy ray_backend module on the default path. + from nemo_skills.pipeline.utils.ray_backend import RayBackend + LOG = logging.getLogger(get_logger_name(__file__)) @@ -36,6 +42,25 @@ def _normalize_backend_config(cluster_config: Dict[str, Any]) -> Dict[str, Any]: return backend_config +# Backend names that select a Ray backend (canonical name plus accepted aliases). +_RAY_BACKEND_NAMES = frozenset({"ray", "kubernetes-ray", "ray-kubernetes", "ray_kubernetes"}) + + +def get_backend_name(cluster_config: Dict[str, Any]) -> str: + """Return the normalized (lowercased) backend name, or '' when none is set. + + Accepts the same string- or dict-form ``backend`` / ``execution_backend`` config as + :func:`get_execution_backend`, so callers never re-implement backend-name parsing + (and never crash on the supported bare-string form, e.g. ``backend: ray``). + """ + return str(_normalize_backend_config(cluster_config).get("name") or "").strip().lower() + + +def is_ray_backend_name(cluster_config: Dict[str, Any]) -> bool: + """True when the configured backend name selects a Ray backend (any alias).""" + return get_backend_name(cluster_config) in _RAY_BACKEND_NAMES + + def _resolve_selector_keys_with_container_map( selectors: Dict[str, Any] | None, containers: Dict[str, Any] | None, @@ -142,15 +167,29 @@ def stop_experiment(self, exp: run.Experiment, only_active: bool = True) -> list # --------------------------------------------------------------------------- -# RayBackend lives in ray_backend.py – imported here for backwards compat. +# RayBackend lives in ray_backend.py. It is imported lazily -- only when a Ray +# backend is actually selected (see get_execution_backend) -- so the default / +# Slurm path never pays to import the ~900-line ray_backend module. A module-level +# __getattr__ (PEP 562) keeps the historical `from ...backends import RayBackend` +# re-export working without eagerly importing the module at package load. # --------------------------------------------------------------------------- -from nemo_skills.pipeline.utils.ray_backend import RayBackend # noqa: E402 # re-exported -# RayBackend is re-exported here (listed in __all__) for backwards-compatible imports. + +def __getattr__(name: str): + if name == "RayBackend": + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + return RayBackend + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") + + +# RayBackend is re-exported here (via __getattr__) for backwards-compatible imports. __all__ = [ "BackendRunOptions", "ExecutionBackend", "RayBackend", + "get_backend_name", + "is_ray_backend_name", "get_execution_backend", "track_stage_tasks", "stop_stage_tasks", @@ -169,6 +208,11 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa backend_name = str(backend_config.get("name") or "").strip().lower() legacy_ray_endpoint = cluster_config.get("ray_endpoint") + # Import RayBackend lazily -- only when a Ray backend is actually selected -- so the + # default / Slurm resolution path never imports the heavy ray_backend module. + if with_ray or backend_name in _RAY_BACKEND_NAMES: + from nemo_skills.pipeline.utils.ray_backend import RayBackend + if not backend_name: if with_ray: return RayBackend( diff --git a/nemo_skills/pipeline/utils/declarative.py b/nemo_skills/pipeline/utils/declarative.py index 9f71373c9b..ddf888efaa 100644 --- a/nemo_skills/pipeline/utils/declarative.py +++ b/nemo_skills/pipeline/utils/declarative.py @@ -31,7 +31,7 @@ run_exp, temporary_env_update, ) -from nemo_skills.pipeline.utils.backends import get_execution_backend +from nemo_skills.pipeline.utils.backends import get_backend_name, get_execution_backend from nemo_skills.pipeline.utils.exp import ( REUSE_CODE_EXP, get_packaging_job_key, @@ -592,7 +592,9 @@ def _rewrite_local_paths(self, script: run.Script) -> run.Script: resolves via the baked image's ``PYTHONPATH``. """ nemo_repo = get_registered_external_repo("nemo_skills") - is_ray = (self.cluster_config.get("backend") or {}).get("name") == "ray" + # Use the shared parser so the supported bare-string form (``backend: ray``) is + # handled too -- a direct ``.get("name")`` raises AttributeError on a string. + is_ray = get_backend_name(self.cluster_config) == "ray" if nemo_repo is None and not is_ray: return script @@ -922,7 +924,12 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: # Ray metadata handling backend = get_execution_backend(cluster_config, with_ray=self.with_ray) - should_use_with_ray_cluster = bool(self.with_ray or getattr(backend, "name", "") == "ray") + # use_with_ray_cluster (embedded Ray-on-Slurm) is only valid on SlurmExecutor, so gate + # on executor == "slurm" -- mirrors add_task so a non-Slurm executor never requests an + # embedded cluster. Precreated Ray Jobs clusters never embed (RayBackend enforces that). + should_use_with_ray_cluster = bool( + self.cluster_config["executor"] == "slurm" and (self.with_ray or getattr(backend, "name", "") == "ray") + ) metadata = backend.stage_metadata( use_with_ray_cluster=should_use_with_ray_cluster, container_image=metadata_container_image, @@ -954,14 +961,16 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: # A run_after naming another experiment whose tasks have already finished # resolves to empty handles and falls through as a bare experiment-name string - # in internal_deps. nemo-run's exp.add asserts every dependency is a job in THIS - # experiment, so drop deps that are not present here -- cross-experiment ordering - # is still honored via external_deps and (on Ray) the queued dep resolver. Only - # filters when exp.jobs is introspectable, so it never over-drops valid handles. + # in internal_deps. Only the Ray backend resolves cross-experiment ordering from + # the queued dep names (external_deps + the Ray queue), so only it needs those + # finished-cross-experiment strings dropped from exp.add(). On the default backend + # we must NOT silently drop them: nemo-run's exp.add asserts every dependency is a + # job in THIS experiment, and that fail-loud behavior is the historical contract. # Only filter when exp.jobs is a concrete list (a real nemo-run experiment); a # mocked or duck-typed exp leaves deps untouched so valid handles are never dropped. + is_ray_backend = getattr(backend, "name", "") == "ray" exp_jobs = getattr(exp, "jobs", None) - if internal_deps and isinstance(exp_jobs, (list, tuple)): + if is_ray_backend and internal_deps and isinstance(exp_jobs, (list, tuple)): known_job_ids = {getattr(job, "id", None) for job in exp_jobs} kept = [] for dep in internal_deps: diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index 0ff73fd476..616f899853 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -28,7 +28,7 @@ from nemo_run.core.execution.slurm import SlurmJobDetails, get_packaging_job_key from torchx.specs.api import AppState -from nemo_skills.pipeline.utils.backends import BackendRunOptions, get_execution_backend +from nemo_skills.pipeline.utils.backends import BackendRunOptions, get_execution_backend, is_ray_backend_name from nemo_skills.pipeline.utils.cluster import ( get_env_variables, get_slurm_timeout_str, @@ -696,6 +696,12 @@ def add_task( command_images = [] executors = [] + # command_images is consumed only by the Ray Jobs queue (queue_ray_job_commands). + # Resolving an image can be expensive (e.g. a `docker inspect` for `dockerfile:` + # specs), so only populate it when a Ray backend is active; the non-Ray path leaves + # it empty (the queue is a no-op there anyway). + collect_command_images = bool(with_ray) or is_ray_backend_name(cluster_config) + def add_server_tasks(): """Append the server (and its executor) for each requested server replica.""" nonlocal het_group @@ -740,7 +746,8 @@ def add_server_tasks(): if cluster_config["executor"] != "slurm" and num_server_tasks > 1: cmd_to_add = f"mpirun --allow-run-as-root -np {num_server_tasks} bash -c {shlex.quote(server_cmd)}" commands.append(cmd_to_add) - command_images.append(resolve_container_image(server_container, cluster_config)) + if collect_command_images: + command_images.append(resolve_container_image(server_container, cluster_config)) executors.append(server_executor) het_group_indices.append(het_group) het_group += 1 @@ -778,7 +785,8 @@ def add_server_tasks(): with temporary_env_update(cluster_config, main_env_updates): cur_cmd = install_packages_wrap(cur_cmd, installation_command) commands.append(cur_cmd) - command_images.append(resolve_container_image(cur_container, cluster_config)) + if collect_command_images: + command_images.append(resolve_container_image(cur_container, cluster_config)) executors.append( get_executor( cluster_config=cluster_config, @@ -828,7 +836,8 @@ def add_server_tasks(): with temporary_env_update(cluster_config, sandbox_env_updates): commands.append(get_sandbox_command(cluster_config)) sandbox_image = sandbox_container or cluster_config["containers"]["sandbox"] - command_images.append(resolve_container_image(sandbox_image, cluster_config)) + if collect_command_images: + command_images.append(resolve_container_image(sandbox_image, cluster_config)) if sandbox_mounts is not None: sandbox_exec_mounts = normalize_mounts_list(sandbox_mounts, allow_rw_mode=True) else: @@ -916,7 +925,14 @@ def add_server_tasks(): backend = get_execution_backend(cluster_config, with_ray=with_ray) first_command_image = command_images[0] if command_images else None - should_use_with_ray_cluster = bool(with_ray or getattr(backend, "name", "") == "ray") + # use_with_ray_cluster requests an EMBEDDED Ray cluster, which nemo-run only supports on + # SlurmExecutor (it asserts otherwise). Gate the whole flag on executor == "slurm": legacy + # with_ray on a non-Slurm executor (and the compat RayBackend it resolves to, whose name is + # also "ray") must NOT request it -- the old code silently no-op'd that combo. A precreated + # Ray Jobs cluster (executor: none) never embeds; RayBackend.stage_metadata enforces that. + should_use_with_ray_cluster = bool( + cluster_config["executor"] == "slurm" and (with_ray or getattr(backend, "name", "") == "ray") + ) metadata = backend.stage_metadata( use_with_ray_cluster=should_use_with_ray_cluster, container_image=first_command_image, @@ -1047,7 +1063,11 @@ def run_exp(exp, cluster_config, sequential=False, dry_run=False): If it is specified, it will be used as is. """ - if "mounts" in cluster_config: + # Skip the live mount check on a dry run: check_remote_mount_directories opens an SSH + # tunnel and stats each source on the cluster, so honoring --dry_run here keeps a dry + # run free of remote I/O (and runnable fully offline), matching historical behavior. + # The backend's start_experiment() handles its own dry-run logging below. + if not dry_run and "mounts" in cluster_config: # Can only check cluster mounts here, not those added to add_task mounts = get_mounts_from_config(cluster_config) mount_sources = [m.split(":")[0] for m in mounts] diff --git a/tests/test_backends.py b/tests/test_backends.py index 94563ce944..17b3c109d9 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -14,7 +14,7 @@ import pytest -from nemo_skills.pipeline.utils.backends import get_execution_backend +from nemo_skills.pipeline.utils.backends import get_backend_name, get_execution_backend, is_ray_backend_name def test_kubernetes_ray_selector_with_image_label_translation(): @@ -361,3 +361,77 @@ def get_job_status(self, job_id): backend._stop_jobs_best_effort(FlakyClient(), ["job-1", "job-2"], reason="poll-failure") assert sorted(stopped) == ["job-1", "job-2"] + + +# --------------------------------------------------------------------------- +# Backend-name parsing helpers (string-form safe) +# --------------------------------------------------------------------------- + + +def test_get_backend_name_handles_string_and_dict_and_alias_forms(): + # Bare-string form must not crash and must normalize like the dict form. + assert get_backend_name({"backend": "ray"}) == "ray" + assert get_backend_name({"backend": {"name": "Ray"}}) == "ray" + assert get_backend_name({"backend": " Kubernetes-Ray "}) == "kubernetes-ray" + assert get_backend_name({"execution_backend": "ray"}) == "ray" + assert get_backend_name({}) == "" + assert get_backend_name({"backend": {}}) == "" + + +def test_is_ray_backend_name_recognizes_aliases_and_rejects_default(): + assert is_ray_backend_name({"backend": "ray"}) is True + assert is_ray_backend_name({"backend": {"name": "kubernetes-ray"}}) is True + assert is_ray_backend_name({"backend": "ray_kubernetes"}) is True + assert is_ray_backend_name({"backend": "ray-kubernetes"}) is True + assert is_ray_backend_name({"backend": "default"}) is False + assert is_ray_backend_name({"backend": "none"}) is False + assert is_ray_backend_name({}) is False + + +def test_default_backend_path_does_not_import_ray_backend(): + """The default / Slurm resolution path must not import the heavy ray_backend module + (the decoupling nit). Verified in a fresh interpreter so it is order-independent, while + the historical ``from ...backends import RayBackend`` re-export still works (lazily). + """ + import subprocess + import sys + + code = ( + "import sys\n" + "import nemo_skills.pipeline.utils.backends as b\n" + "assert 'nemo_skills.pipeline.utils.ray_backend' not in sys.modules, 'imported at module load'\n" + "b.get_execution_backend({'executor': 'slurm'})\n" + "b.get_execution_backend({'executor': 'slurm', 'backend': {'name': 'default'}})\n" + "assert 'nemo_skills.pipeline.utils.ray_backend' not in sys.modules, 'imported on default path'\n" + "from nemo_skills.pipeline.utils.backends import RayBackend\n" + "assert RayBackend.__name__ == 'RayBackend'\n" + "assert 'nemo_skills.pipeline.utils.ray_backend' in sys.modules, 're-export should import it'\n" + "print('OK')\n" + ) + result = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True) + assert result.returncode == 0, result.stderr + assert "OK" in result.stdout + + +# --------------------------------------------------------------------------- +# run_exp must honor --dry_run before any live remote I/O (non-Ray regression guard) +# --------------------------------------------------------------------------- + + +def test_run_exp_dry_run_skips_remote_mount_check(monkeypatch): + """A --dry_run must not perform live remote I/O on the default (non-Ray) path: the mount + check opens an SSH tunnel and stats each source on the cluster. It must be skipped on a + dry run so a dry run stays offline-safe (guards the regression where the early dry-run + short-circuit was removed).""" + from unittest.mock import MagicMock + + from nemo_skills.pipeline.utils import exp as exp_mod + + calls = [] + monkeypatch.setattr(exp_mod, "get_mounts_from_config", lambda cc: cc["mounts"]) + monkeypatch.setattr(exp_mod, "check_remote_mount_directories", lambda *a, **k: calls.append(("check", a))) + + cluster_config = {"executor": "slurm", "mounts": ["/src:/dst"]} + # Default backend + slurm + dry_run: must return without touching the cluster. + exp_mod.run_exp(MagicMock(), cluster_config, dry_run=True) + assert calls == [] diff --git a/tests/test_declarative_pipeline.py b/tests/test_declarative_pipeline.py index 6cae8f682f..3ad2be4b24 100644 --- a/tests/test_declarative_pipeline.py +++ b/tests/test_declarative_pipeline.py @@ -722,12 +722,14 @@ def mock_get_executor(**kwargs): call2_kwargs = mock_exp.add.call_args_list[1][1] assert call2_kwargs["dependencies"] == ["task_handle_1"] - def test_finished_cross_experiment_dep_filtered_from_exp_add(self): - """A run_after naming another experiment whose tasks already finished resolves to - empty handles and falls through as a bare experiment-name string in internal_deps - (the _reuse_exp path). It must be dropped from exp.add (nemo-run asserts every dep - is a job in THIS experiment) while same-experiment handles are kept. Reproduces the - bug the legacy cross-experiment dependency patch fixed, now handled natively. + def test_finished_cross_experiment_dep_filtered_from_exp_add_on_ray_backend(self): + """On the RAY backend, a run_after naming another experiment whose tasks already + finished resolves to empty handles and falls through as a bare experiment-name + string in internal_deps (the _reuse_exp path). The Ray backend resolves + cross-experiment ordering from the queued dep names (not nemo-run handles), so that + finished-cross-exp string is dropped from exp.add while same-experiment handles are + kept. Reproduces the bug the legacy cross-experiment dependency patch fixed, now + handled natively. The default backend must NOT drop it -- see the sibling test. """ import nemo_run as run @@ -761,6 +763,9 @@ def mock_get_executor(**kwargs): with patch("nemo_skills.pipeline.utils.declarative.run_exp"): cluster_config = { "executor": "slurm", + # Ray backend (no dashboard) -> backend.name == "ray", so the + # finished-cross-exp drop block is active. + "backend": {"name": "ray"}, "containers": {"nemo-skills": "test/container"}, "account": "test", "env_vars": {"HF_HOME": "/mounted/hf_home"}, @@ -797,6 +802,80 @@ def mock_get_executor(**kwargs): # job2: same-experiment handle kept, finished cross-exp string dropped. assert mock_exp.add.call_args_list[1][1]["dependencies"] == ["task_handle_1"] + def test_finished_cross_experiment_dep_kept_on_default_backend(self): + """The default (non-Ray) backend must NOT silently drop a finished cross-experiment + string dep. nemo-run's real exp.add asserts every dependency is a job in THIS + experiment and raises "Dependency not found" -- that fail-loud behavior is the + historical contract, so the string must reach exp.add unchanged (here exp.add is + mocked, so we assert the string is passed through rather than silently dropped). + """ + import nemo_run as run + + with patch("nemo_skills.pipeline.utils.declarative.get_exp_handles") as mock_get_handles: + mock_get_handles.return_value = [] + + with patch("nemo_skills.pipeline.utils.declarative.get_exp") as mock_get_exp: + mock_exp = MagicMock(spec=run.Experiment) + mock_exp.__enter__ = MagicMock(return_value=mock_exp) + mock_exp.__exit__ = MagicMock(return_value=False) + mock_exp.jobs = [] + + def fake_add(*args, **kwargs): + handle = f"task_handle_{len(mock_exp.jobs) + 1}" + job = MagicMock() + job.id = handle + mock_exp.jobs.append(job) + return handle + + mock_exp.add = MagicMock(side_effect=fake_add) + mock_get_exp.return_value = mock_exp + + def mock_get_executor(**kwargs): + mock_executor = MagicMock() + mock_executor.packager = MagicMock() + return mock_executor + + with patch("nemo_skills.pipeline.utils.declarative.get_executor", side_effect=mock_get_executor): + with patch("nemo_skills.pipeline.utils.declarative.run_exp"): + cluster_config = { + # No backend key -> default ExecutionBackend (the non-Ray path). + "executor": "slurm", + "containers": {"nemo-skills": "test/container"}, + "account": "test", + "env_vars": {"HF_HOME": "/mounted/hf_home"}, + "mounts": ["/mounted/hf_home:/mounted/hf_home"], + } + + cmd1 = make_command(inline="echo job1", name="job1") + group1 = CommandGroup(commands=[cmd1], name="group1", log_dir="/tmp/logs") + cmd2 = make_command(inline="echo job2", name="job2") + group2 = CommandGroup(commands=[cmd2], name="group2", log_dir="/tmp/logs") + + job1_spec = {"name": "job1", "group": group1} + job2_spec = { + "name": "job2", + "group": group2, + "dependencies": [job1_spec, "finished_external_experiment"], + } + + pipeline = Pipeline( + name="test_pipeline", + cluster_config=cluster_config, + jobs=[job1_spec, job2_spec], + skip_hf_home_check=True, + reuse_code=False, + ) + + pipeline.run(dry_run=True, _reuse_exp=mock_exp) + + assert mock_exp.add.call_count == 2 + assert mock_exp.add.call_args_list[0][1]["dependencies"] is None + # job2: the finished cross-exp string is NOT dropped on the default backend. + deps = mock_exp.add.call_args_list[1][1]["dependencies"] + assert deps is not None + assert "task_handle_1" in deps + assert "finished_external_experiment" in deps + def test_run_after_dependencies_across_experiments(self, tmp_path): """Test that run_after dependencies work when chaining multiple generate/run_cmd calls. diff --git a/tests/test_pipeline_utils.py b/tests/test_pipeline_utils.py index bb2827d48f..7e0460857e 100644 --- a/tests/test_pipeline_utils.py +++ b/tests/test_pipeline_utils.py @@ -457,3 +457,90 @@ def test_add_task_sandbox_mounts_override_keep_mounts_true(mock_port, mock_get_e ) assert mock_get_executor.call_args_list[-1].kwargs["mounts"] == ["/host/data:/sandbox/data:ro"] + + +@patch("nemo_skills.pipeline.utils.exp.get_executor") +@patch("nemo_skills.pipeline.utils.exp.get_free_port", return_value=12345) +def test_add_task_with_ray_on_non_slurm_does_not_request_embedded_cluster(mock_port, mock_get_executor): + """Legacy with_ray=True on a non-Slurm executor must NOT attach use_with_ray_cluster. + + nemo-run only supports an embedded Ray cluster on SlurmExecutor (it asserts otherwise). The + legacy compat path resolves with_ray to a RayBackend whose name is "ray", so gating only the + with_ray term is insufficient -- the flag must be gated on executor == "slurm". The old code + silently no-op'd this combo; this guards the regression where it began requesting embedding. + """ + from types import SimpleNamespace + + from nemo_skills.pipeline.utils.exp import add_task + + mock_get_executor.return_value = MagicMock() + captured = {} + + def _add(script, **kwargs): + captured["script"] = script + return "task_handle" + + exp = SimpleNamespace(add=_add) + cluster_config = {"executor": "local", "containers": {"sandbox": "sandbox:latest"}} + + add_task( + exp=exp, + cmd="echo hello", + task_name="test-task", + cluster_config=cluster_config, + container="main:latest", + log_dir="/tmp/logs", + with_ray=True, + skip_hf_home_check=True, + reuse_code=False, + ) + + metadata = getattr(captured["script"], "metadata", None) or {} + assert "use_with_ray_cluster" not in metadata, ( + f"with_ray=True on a non-Slurm executor must not request an embedded Ray cluster; got {metadata}" + ) + + +@patch("nemo_skills.pipeline.utils.exp.resolve_container_image", return_value="resolved:image") +@patch("nemo_skills.pipeline.utils.exp.get_executor") +@patch("nemo_skills.pipeline.utils.exp.get_free_port", return_value=12345) +def test_add_task_resolves_command_images_only_when_ray_backend_active(mock_port, mock_get_executor, mock_resolve): + """command_images is consumed only by the Ray Jobs queue, and resolving an image can be + expensive (a docker inspect for dockerfile: specs). With get_executor mocked, the only + remaining caller of resolve_container_image is the (guarded) command_images appends -- so it + must not be called on the non-Ray path and must be called when a Ray backend is active. + """ + from types import SimpleNamespace + + from nemo_skills.pipeline.utils.exp import add_task + + mock_get_executor.return_value = MagicMock() + cluster_config = {"executor": "local", "containers": {"sandbox": "sandbox:latest"}} + + # Non-Ray backend: no command-image resolution. + add_task( + exp=SimpleNamespace(add=MagicMock(return_value="h")), + cmd="echo hello", + task_name="t", + cluster_config=cluster_config, + container="main:latest", + log_dir="/tmp/logs", + skip_hf_home_check=True, + reuse_code=False, + ) + assert mock_resolve.call_count == 0, "non-Ray add_task must not resolve command images" + + # Ray-active path (legacy with_ray): images ARE resolved for the Ray queue. + mock_resolve.reset_mock() + add_task( + exp=SimpleNamespace(add=MagicMock(return_value="h")), + cmd="echo hello", + task_name="t", + cluster_config=cluster_config, + container="main:latest", + log_dir="/tmp/logs", + with_ray=True, + skip_hf_home_check=True, + reuse_code=False, + ) + assert mock_resolve.call_count >= 1, "Ray-active add_task must resolve command images for the queue" From a0760f860c4b84496e9dd8a0965499a0972498e1 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 30 Jun 2026 10:23:40 -0400 Subject: [PATCH 25/46] test(ray): assert use_with_ray_cluster key absence, not falsiness (review) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CodeRabbit: the precreated/non-precreated stage_metadata assertions used `not (metadata or {}).get("use_with_ray_cluster")`, which also passes if the key is present as False/None — missing the regression the test names guard. Tighten to assert the key is absent from the metadata dict. Keep the `or {}` guard since the base ExecutionBackend.stage_metadata returns None by contract. Signed-off-by: Nick Gupta --- tests/test_backends.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_backends.py b/tests/test_backends.py index 17b3c109d9..670dfb3e5a 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -175,7 +175,7 @@ def test_dashboard_only_ray_config_is_precreated_not_embedded(): assert backend.precreated_cluster is True metadata = backend.stage_metadata(container_image="nvcr.io/nvidia/pytorch:25.02-py3") - assert not (metadata or {}).get("use_with_ray_cluster") + assert "use_with_ray_cluster" not in (metadata or {}) def test_dashboard_only_precreated_preflight_does_not_require_endpoint(): @@ -214,7 +214,7 @@ def test_non_precreated_ray_honors_use_with_ray_cluster_flag(): assert backend.precreated_cluster is False assert backend.stage_metadata(use_with_ray_cluster=True).get("use_with_ray_cluster") is True - assert not (backend.stage_metadata(use_with_ray_cluster=False) or {}).get("use_with_ray_cluster") + assert "use_with_ray_cluster" not in (backend.stage_metadata(use_with_ray_cluster=False) or {}) def test_ray_executor_none_without_dashboard_raises(): From 30782d92cea6a389a8ae61cabc9959ffef063153 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 4 Aug 2026 18:27:40 -0400 Subject: [PATCH 26/46] fix(ray): avoid duplicate logs in reused experiments Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/exp.py | 5 +++-- tests/test_pipeline_utils.py | 14 ++++++++++++++ 2 files changed, 17 insertions(+), 2 deletions(-) diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index 616f899853..444cb3c665 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -1114,11 +1114,12 @@ def run_exp(exp, cluster_config, sequential=False, dry_run=False): def get_exp(expname, cluster_config, _reuse_exp=None): + # nemo-run defines the root handlers, so remove ours before creating or + # reusing an experiment to avoid duplicate logs from propagated records. + remove_handlers() # Use existing experiment if provided, otherwise create a new one if _reuse_exp: return contextlib.nullcontext(_reuse_exp) - # nemo-run redefines the handlers, so removing ours to avoid duplicate logs - remove_handlers() if cluster_config["executor"] == "slurm": return run.Experiment( expname, diff --git a/tests/test_pipeline_utils.py b/tests/test_pipeline_utils.py index 7e0460857e..83001bf1ab 100644 --- a/tests/test_pipeline_utils.py +++ b/tests/test_pipeline_utils.py @@ -12,6 +12,7 @@ # See the License for the specific language governing permissions and # limitations under the License. +import logging import os import tempfile from unittest.mock import MagicMock, patch @@ -19,6 +20,7 @@ import pytest from nemo_skills.pipeline.utils.declarative import Command +from nemo_skills.pipeline.utils.exp import get_exp from nemo_skills.pipeline.utils.generation import ( get_chunked_rs_filename, get_expected_done_files, @@ -37,6 +39,18 @@ def create_done_files(output_dir, seed_chunk_pairs): f.write("") +def test_get_exp_reuse_removes_nemo_skills_handlers(monkeypatch): + """A nested pipeline must not retain a handler beside nemo-run's root handler.""" + logger = logging.getLogger("nemo_skills") + monkeypatch.setattr(logger, "handlers", [logging.NullHandler()]) + reused_exp = object() + + with get_exp("nested", {"executor": "local"}, _reuse_exp=reused_exp) as exp: + assert exp is reused_exp + + assert logger.handlers == [] + + def test_get_chunked_rs_filename(): """Test filename generation with different parameters.""" assert get_chunked_rs_filename("/tmp/output", random_seed=42) == "/tmp/output/output-rs42.jsonl" From bb09982eb41e2c09ab5964d814a36c59c9997eaf Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 4 Aug 2026 20:46:25 -0400 Subject: [PATCH 27/46] fix: isolate NLTK data download during image build Signed-off-by: Nick Gupta --- dockerfiles/Dockerfile.nemo-skills | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/dockerfiles/Dockerfile.nemo-skills b/dockerfiles/Dockerfile.nemo-skills index 0819e58d51..4e19502149 100644 --- a/dockerfiles/Dockerfile.nemo-skills +++ b/dockerfiles/Dockerfile.nemo-skills @@ -60,7 +60,7 @@ COPY dockerfiles/ifbench.patch /opt/benchmarks/IFBench/ifbench.patch RUN cd /opt/benchmarks/IFBench && git apply ifbench.patch RUN pip install langdetect absl-py immutabledict nltk ipython && \ - python -c "import nltk; from spacy.cli import download; nltk.download('punkt'); nltk.download('punkt_tab'); \ + python -I -c "import nltk; from spacy.cli import download; nltk.download('punkt'); nltk.download('punkt_tab'); \ nltk.download('stopwords'); nltk.download('averaged_perceptron_tagger_eng'); download('en_core_web_sm')" # we aren't copying main nemo_skills folder as it will always be mounted from host From 1fea3376cfd386b0e27eec7bde660885e0f26f1c Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 4 Aug 2026 20:53:31 -0400 Subject: [PATCH 28/46] fix: pin NLTK for the Python 3.10 image Signed-off-by: Nick Gupta --- dockerfiles/Dockerfile.nemo-skills | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/dockerfiles/Dockerfile.nemo-skills b/dockerfiles/Dockerfile.nemo-skills index 4e19502149..f3d92c9f17 100644 --- a/dockerfiles/Dockerfile.nemo-skills +++ b/dockerfiles/Dockerfile.nemo-skills @@ -59,7 +59,8 @@ RUN cd ${IFBENCH_DIR} && pip install -r requirements.txt COPY dockerfiles/ifbench.patch /opt/benchmarks/IFBench/ifbench.patch RUN cd /opt/benchmarks/IFBench && git apply ifbench.patch -RUN pip install langdetect absl-py immutabledict nltk ipython && \ +# NLTK 3.10's import-safety hook is incompatible with this image's Python 3.10 runtime. +RUN pip install langdetect absl-py immutabledict "nltk==3.9.2" ipython && \ python -I -c "import nltk; from spacy.cli import download; nltk.download('punkt'); nltk.download('punkt_tab'); \ nltk.download('stopwords'); nltk.download('averaged_perceptron_tagger_eng'); download('en_core_web_sm')" From 0f1812e2df4266e43fc00b80a13fe75e7398df16 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Tue, 4 Aug 2026 22:01:10 -0400 Subject: [PATCH 29/46] fix(ray): preserve Slurm execution semantics Signed-off-by: Nick Gupta --- nemo_skills/pipeline/utils/backends.py | 22 +++++++-- nemo_skills/pipeline/utils/declarative.py | 40 +++++++-------- nemo_skills/pipeline/utils/exp.py | 59 ++++++++++++++--------- tests/test_backends.py | 4 ++ tests/test_declarative_pipeline.py | 10 ++-- tests/test_pipeline_utils.py | 59 ++++++++++++++++++++--- 6 files changed, 138 insertions(+), 56 deletions(-) diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index 851531ed8e..3af8d27e4a 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -61,6 +61,11 @@ def is_ray_backend_name(cluster_config: Dict[str, Any]) -> bool: return get_backend_name(cluster_config) in _RAY_BACKEND_NAMES +def is_ray_jobs_backend(backend: Any) -> bool: + """True only for a resolved Ray backend that can submit through the Jobs API.""" + return getattr(backend, "name", "") == "ray" and bool(getattr(backend, "dashboard_url", None)) + + def _resolve_selector_keys_with_container_map( selectors: Dict[str, Any] | None, containers: Dict[str, Any] | None, @@ -190,6 +195,7 @@ def __getattr__(name: str): "RayBackend", "get_backend_name", "is_ray_backend_name", + "is_ray_jobs_backend", "get_execution_backend", "track_stage_tasks", "stop_stage_tasks", @@ -201,25 +207,31 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa Resolution priority: 1. Explicit backend/ execution_backend in cluster config - 2. Legacy with_ray compatibility flag - 3. Default backend + 2. Legacy non-Slurm Ray endpoint compatibility + 3. Default backend (including embedded Ray-on-Slurm via with_ray=True) """ backend_config = _normalize_backend_config(cluster_config) backend_name = str(backend_config.get("name") or "").strip().lower() legacy_ray_endpoint = cluster_config.get("ray_endpoint") + executor_kind = str(cluster_config.get("executor") or "").strip().lower() + legacy_precreated_ray = bool(with_ray and legacy_ray_endpoint and executor_kind != "slurm") # Import RayBackend lazily -- only when a Ray backend is actually selected -- so the # default / Slurm resolution path never imports the heavy ray_backend module. - if with_ray or backend_name in _RAY_BACKEND_NAMES: + if legacy_precreated_ray or backend_name in _RAY_BACKEND_NAMES: from nemo_skills.pipeline.utils.ray_backend import RayBackend if not backend_name: - if with_ray: + if legacy_precreated_ray: return RayBackend( endpoint=legacy_ray_endpoint, - precreated_cluster=bool(legacy_ray_endpoint), + precreated_cluster=True, env_vars=get_env_variables(cluster_config), ) + # `with_ray=True` on Slurm is the historical embedded-Ray mode. It is + # metadata on a normal Slurm task, not selection of the Ray Jobs API + # backend. Keeping the default backend here avoids importing Ray, + # resolving queue images, or filtering dependencies on that path. return ExecutionBackend() if backend_name in {"default", "none"}: diff --git a/nemo_skills/pipeline/utils/declarative.py b/nemo_skills/pipeline/utils/declarative.py index ddf888efaa..1dc35f2671 100644 --- a/nemo_skills/pipeline/utils/declarative.py +++ b/nemo_skills/pipeline/utils/declarative.py @@ -31,7 +31,7 @@ run_exp, temporary_env_update, ) -from nemo_skills.pipeline.utils.backends import get_backend_name, get_execution_backend +from nemo_skills.pipeline.utils.backends import get_backend_name, get_execution_backend, is_ray_jobs_backend from nemo_skills.pipeline.utils.exp import ( REUSE_CODE_EXP, get_packaging_job_key, @@ -924,6 +924,7 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: # Ray metadata handling backend = get_execution_backend(cluster_config, with_ray=self.with_ray) + ray_jobs_active = is_ray_jobs_backend(backend) # use_with_ray_cluster (embedded Ray-on-Slurm) is only valid on SlurmExecutor, so gate # on executor == "slurm" -- mirrors add_task so a non-Slurm executor never requests an # embedded cluster. Precreated Ray Jobs clusters never embed (RayBackend enforces that). @@ -937,27 +938,29 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: ray_queue_commands = [] ray_queue_images = [] - for script, executor in zip(scripts, executors): - if not isinstance(script.inline, str): - continue - ray_queue_commands.append(script.inline) - ray_queue_images.append(getattr(executor, "container_image", None)) + if ray_jobs_active: + for script, executor in zip(scripts, executors): + if not isinstance(script.inline, str): + continue + ray_queue_commands.append(script.inline) + ray_queue_images.append(getattr(executor, "container_image", None)) # Forward both internal (same-experiment) and external (cross-experiment # run_after) dependencies. Dropping external deps would let Ray jobs submit # before their prerequisites finish, since Ray ordering is resolved from # the queued dep names rather than the nemo-run executor. - queue_ray_job_commands( - exp=exp, - backend=backend, - commands=ray_queue_commands, - command_images=ray_queue_images, - task_name=groups[0].name, - log_dir=log_dir, - task_dependencies=internal_deps, - external_dependencies=external_deps, - should_use_with_ray_cluster=should_use_with_ray_cluster, - ) + if ray_jobs_active: + queue_ray_job_commands( + exp=exp, + backend=backend, + commands=ray_queue_commands, + command_images=ray_queue_images, + task_name=groups[0].name, + log_dir=log_dir, + task_dependencies=internal_deps, + external_dependencies=external_deps, + should_use_with_ray_cluster=should_use_with_ray_cluster, + ) # A run_after naming another experiment whose tasks have already finished # resolves to empty handles and falls through as a bare experiment-name string @@ -968,9 +971,8 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: # job in THIS experiment, and that fail-loud behavior is the historical contract. # Only filter when exp.jobs is a concrete list (a real nemo-run experiment); a # mocked or duck-typed exp leaves deps untouched so valid handles are never dropped. - is_ray_backend = getattr(backend, "name", "") == "ray" exp_jobs = getattr(exp, "jobs", None) - if is_ray_backend and internal_deps and isinstance(exp_jobs, (list, tuple)): + if ray_jobs_active and internal_deps and isinstance(exp_jobs, (list, tuple)): known_job_ids = {getattr(job, "id", None) for job in exp_jobs} kept = [] for dep in internal_deps: diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index 444cb3c665..a3abfc025e 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -28,7 +28,12 @@ from nemo_run.core.execution.slurm import SlurmJobDetails, get_packaging_job_key from torchx.specs.api import AppState -from nemo_skills.pipeline.utils.backends import BackendRunOptions, get_execution_backend, is_ray_backend_name +from nemo_skills.pipeline.utils.backends import ( + BackendRunOptions, + get_execution_backend, + is_ray_backend_name, + is_ray_jobs_backend, +) from nemo_skills.pipeline.utils.cluster import ( get_env_variables, get_slurm_timeout_str, @@ -696,11 +701,14 @@ def add_task( command_images = [] executors = [] + backend = get_execution_backend(cluster_config, with_ray=with_ray) + ray_jobs_active = is_ray_jobs_backend(backend) + # command_images is consumed only by the Ray Jobs queue (queue_ray_job_commands). # Resolving an image can be expensive (e.g. a `docker inspect` for `dockerfile:` # specs), so only populate it when a Ray backend is active; the non-Ray path leaves # it empty (the queue is a no-op there anyway). - collect_command_images = bool(with_ray) or is_ray_backend_name(cluster_config) + collect_command_images = ray_jobs_active def add_server_tasks(): """Append the server (and its executor) for each requested server replica.""" @@ -923,7 +931,6 @@ def add_server_tasks(): ) commands[idx] = commands[idx].replace("/nemo_run/code", "./") - backend = get_execution_backend(cluster_config, with_ray=with_ray) first_command_image = command_images[0] if command_images else None # use_with_ray_cluster requests an EMBEDDED Ray cluster, which nemo-run only supports on # SlurmExecutor (it asserts otherwise). Gate the whole flag on executor == "slurm": legacy @@ -937,27 +944,29 @@ def add_server_tasks(): use_with_ray_cluster=should_use_with_ray_cluster, container_image=first_command_image, ) - LOG.info( - "Execution backend resolved to '%s' (dashboard_url=%s).", - getattr(backend, "name", "unknown"), - getattr(backend, "dashboard_url", None), - ) + if getattr(backend, "name", "default") != "default": + LOG.info( + "Execution backend resolved to '%s' (dashboard_url=%s).", + getattr(backend, "name", "unknown"), + getattr(backend, "dashboard_url", None), + ) # For Ray Jobs API mode, queue commands on the experiment and let the backend # submit/track/cancel them centrally in start_experiment(). `dependencies` # holds the resolved external run_after handles; forward them too so the Ray # queue waits on cross-experiment prerequisites, not just same-experiment ones. - queue_ray_job_commands( - exp=exp, - backend=backend, - commands=commands, - command_images=command_images, - task_name=task_name, - log_dir=log_dir, - task_dependencies=task_dependencies, - external_dependencies=dependencies, - should_use_with_ray_cluster=should_use_with_ray_cluster, - ) + if ray_jobs_active: + queue_ray_job_commands( + exp=exp, + backend=backend, + commands=commands, + command_images=command_images, + task_name=task_name, + log_dir=log_dir, + task_dependencies=task_dependencies, + external_dependencies=dependencies, + should_use_with_ray_cluster=should_use_with_ray_cluster, + ) if not task_dependencies: # empty list task_dependencies = None @@ -1114,12 +1123,18 @@ def run_exp(exp, cluster_config, sequential=False, dry_run=False): def get_exp(expname, cluster_config, _reuse_exp=None): - # nemo-run defines the root handlers, so remove ours before creating or - # reusing an experiment to avoid duplicate logs from propagated records. - remove_handlers() # Use existing experiment if provided, otherwise create a new one if _reuse_exp: + # Reused Ray Jobs experiments otherwise retain our handler beside + # nemo-run's root handler and duplicate every record. A reused Slurm + # experiment historically kept its handler, so do not mutate it. + if is_ray_backend_name(cluster_config): + backend = get_execution_backend(cluster_config) + if is_ray_jobs_backend(backend): + remove_handlers() return contextlib.nullcontext(_reuse_exp) + # nemo-run redefines the handlers, so removing ours avoids duplicate logs. + remove_handlers() if cluster_config["executor"] == "slurm": return run.Experiment( expname, diff --git a/tests/test_backends.py b/tests/test_backends.py index 670dfb3e5a..a9f3fbff09 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -402,6 +402,10 @@ def test_default_backend_path_does_not_import_ray_backend(): "assert 'nemo_skills.pipeline.utils.ray_backend' not in sys.modules, 'imported at module load'\n" "b.get_execution_backend({'executor': 'slurm'})\n" "b.get_execution_backend({'executor': 'slurm', 'backend': {'name': 'default'}})\n" + "embedded = b.get_execution_backend({'executor': 'slurm'}, with_ray=True)\n" + "assert embedded.name == 'default'\n" + "assert embedded.stage_metadata(use_with_ray_cluster=True) == " + "{'use_with_ray_cluster': True}\n" "assert 'nemo_skills.pipeline.utils.ray_backend' not in sys.modules, 'imported on default path'\n" "from nemo_skills.pipeline.utils.backends import RayBackend\n" "assert RayBackend.__name__ == 'RayBackend'\n" diff --git a/tests/test_declarative_pipeline.py b/tests/test_declarative_pipeline.py index 3ad2be4b24..aaad12d9d1 100644 --- a/tests/test_declarative_pipeline.py +++ b/tests/test_declarative_pipeline.py @@ -763,9 +763,13 @@ def mock_get_executor(**kwargs): with patch("nemo_skills.pipeline.utils.declarative.run_exp"): cluster_config = { "executor": "slurm", - # Ray backend (no dashboard) -> backend.name == "ray", so the - # finished-cross-exp drop block is active. - "backend": {"name": "ray"}, + # A real Ray Jobs backend owns cross-experiment ordering + # through its queue, so the finished dependency string is + # filtered from nemo-run's local dependency list. + "backend": { + "name": "ray", + "dashboard_url": "http://ray-head:8265", + }, "containers": {"nemo-skills": "test/container"}, "account": "test", "env_vars": {"HF_HOME": "/mounted/hf_home"}, diff --git a/tests/test_pipeline_utils.py b/tests/test_pipeline_utils.py index 83001bf1ab..fdda6cf03f 100644 --- a/tests/test_pipeline_utils.py +++ b/tests/test_pipeline_utils.py @@ -39,13 +39,32 @@ def create_done_files(output_dir, seed_chunk_pairs): f.write("") -def test_get_exp_reuse_removes_nemo_skills_handlers(monkeypatch): - """A nested pipeline must not retain a handler beside nemo-run's root handler.""" +def test_get_exp_reuse_preserves_slurm_handlers(monkeypatch): + """Reusing a Slurm experiment preserves the historical logger state.""" logger = logging.getLogger("nemo_skills") monkeypatch.setattr(logger, "handlers", [logging.NullHandler()]) reused_exp = object() - with get_exp("nested", {"executor": "local"}, _reuse_exp=reused_exp) as exp: + with get_exp("nested", {"executor": "slurm"}, _reuse_exp=reused_exp) as exp: + assert exp is reused_exp + + assert len(logger.handlers) == 1 + + +def test_get_exp_reuse_removes_ray_jobs_handlers(monkeypatch): + """A nested Ray Jobs pipeline must not duplicate propagated log records.""" + logger = logging.getLogger("nemo_skills") + monkeypatch.setattr(logger, "handlers", [logging.NullHandler()]) + reused_exp = object() + cluster_config = { + # A pre-created Ray Jobs backend can use Slurm as the surrounding + # scheduler; backend selection, not the executor label, controls the + # duplicate-handler cleanup. + "executor": "slurm", + "backend": {"name": "ray", "dashboard_url": "http://ray-head:8265"}, + } + + with get_exp("nested", cluster_config, _reuse_exp=reused_exp) as exp: assert exp is reused_exp assert logger.handlers == [] @@ -544,17 +563,43 @@ def test_add_task_resolves_command_images_only_when_ray_backend_active(mock_port ) assert mock_resolve.call_count == 0, "non-Ray add_task must not resolve command images" - # Ray-active path (legacy with_ray): images ARE resolved for the Ray queue. + # Legacy embedded Ray-on-Slurm is not the Jobs API backend and must not + # resolve images or prepare a queue. mock_resolve.reset_mock() + captured = {} + + def _add(script, **kwargs): + captured["metadata"] = script.metadata + return "h" + add_task( - exp=SimpleNamespace(add=MagicMock(return_value="h")), + exp=SimpleNamespace(add=_add), cmd="echo hello", task_name="t", - cluster_config=cluster_config, + cluster_config={"executor": "slurm", "containers": {"sandbox": "sandbox:latest"}}, container="main:latest", log_dir="/tmp/logs", with_ray=True, skip_hf_home_check=True, reuse_code=False, ) - assert mock_resolve.call_count >= 1, "Ray-active add_task must resolve command images for the queue" + assert mock_resolve.call_count == 0 + assert captured["metadata"] == {"use_with_ray_cluster": True} + + # An explicit Ray Jobs backend does need the resolved image for its queue. + ray_jobs_config = { + "executor": "none", + "containers": {"sandbox": "sandbox:latest"}, + "backend": {"name": "ray", "dashboard_url": "http://ray-head:8265"}, + } + add_task( + exp=SimpleNamespace(add=MagicMock(return_value="h"), jobs=[]), + cmd="echo hello", + task_name="t", + cluster_config=ray_jobs_config, + container="main:latest", + log_dir="/tmp/logs", + skip_hf_home_check=True, + reuse_code=False, + ) + assert mock_resolve.call_count >= 1, "Ray Jobs add_task must resolve command images for the queue" From f25abc37e99bc8b5be4cefbf0132582c9ae36c23 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Thu, 9 Jul 2026 11:30:10 -0400 Subject: [PATCH 30/46] fix(security): floor litellm/wandb/lxml past known CVEs; relax stale click cap MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - litellm[caching] 1.83.14 -> 1.84.10: GHSA-4xpc-pv4p-pm3w (Critical) — 1.83.x leaks the API key to an arbitrary attacker-controlled Host header; fixed in 1.84.0. Minimal exact-pin jump; resolves with the existing httpx[http2]>=0.28.1 override (litellm 1.84.10 needs httpx>=0.28.0). - wandb -> >=0.27.1: the bundled wandb-core Go binary in older wheels ships golang.org/x/crypto 0.50.0 + Go 1.26.2 stdlib with 7 Critical / 13 High CVEs (incl. GHSA-x527-x647-q7gg et al.); 0.27.1 is the first release embedding patched x/crypto 0.52.0 (verified on both arches). - click < 8.2.0 cap removed + typer >= 0.16: the cap guarded against the typer/click-8.2 make_metavar break (ai-dynamo/dynamo#1039, closed 2025-06-26, fixed in typer >=0.16); wandb>=0.27.1 requires click>=8.2, and requires-python >=3.10 satisfies click 8.2's floor. - lxml -> >=6.1.0 (stem extra): GHSA-vfmq-68hx-4jfw (High). Validation: uv pip compile of core+pipeline (py3.10, with the pyproject overrides) resolves cleanly — litellm 1.84.10 / wandb 0.28.0 / click 8.4.2 / typer 0.26.8 / httpx 0.28.1; stem extra resolves with lxml 6.1.1. Runtime smoke on the resolved set: litellm/wandb import clean; typer --help rendering (the exact make_metavar crash path) passes under click 8.4. Signed-off-by: Nick Gupta --- requirements/stem.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements/stem.txt b/requirements/stem.txt index ad993f3bcb..6e84f85dac 100644 --- a/requirements/stem.txt +++ b/requirements/stem.txt @@ -77,7 +77,7 @@ lie LIEGenTools lifelines lingpy -lxml +lxml>=6.1.0 # fixes GHSA-vfmq-68hx-4jfw (High) matplotlib mendeleev mido From e16936ecb473725773d694a1f788b9d514bcc549 Mon Sep 17 00:00:00 2001 From: "coderabbitai[bot]" <136622811+coderabbitai[bot]@users.noreply.github.com> Date: Thu, 9 Jul 2026 15:49:01 +0000 Subject: [PATCH 31/46] =?UTF-8?q?=F0=9F=93=9D=20CodeRabbit=20Chat:=20Add?= =?UTF-8?q?=20unit=20tests=20for=20PR=20changes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tests/test_dependency_pins.py | 267 ++++++++++++++++++++++++++++++++++ 1 file changed, 267 insertions(+) create mode 100644 tests/test_dependency_pins.py diff --git a/tests/test_dependency_pins.py b/tests/test_dependency_pins.py new file mode 100644 index 0000000000..0986d94eb9 --- /dev/null +++ b/tests/test_dependency_pins.py @@ -0,0 +1,267 @@ +# Copyright (c) 2025, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Regression tests for security-motivated dependency pins. + +This suite guards the version floors/pins introduced to remediate known +CVEs/GHSAs (litellm host-header key leak, wandb-core x/crypto bundle, lxml +GHSA-vfmq-68hx-4jfw) as well as the click/typer compatibility fix that was +required to unblock the wandb>=0.27.1 upgrade. It does *not* attempt to +validate the full dependency graph -- only the specific lines that were +touched by the security-pin PR, plus enough surrounding structure to catch +someone accidentally reverting/loosening them in the future. +""" + +import pathlib +import re + +import pytest +from packaging.requirements import InvalidRequirement, Requirement +from packaging.version import Version + +try: + import tomllib +except ModuleNotFoundError: # pragma: no cover - py<3.11 fallback + try: + import tomli as tomllib + except ModuleNotFoundError: + tomllib = None + +REPO_ROOT = pathlib.Path(__file__).parent.parent +CORE_REQUIREMENTS = REPO_ROOT / "core" / "requirements.txt" +PIPELINE_REQUIREMENTS = REPO_ROOT / "requirements" / "pipeline.txt" +STEM_REQUIREMENTS = REPO_ROOT / "requirements" / "stem.txt" +PYPROJECT_TOML = REPO_ROOT / "pyproject.toml" + + +def _iter_non_comment_lines(path: pathlib.Path): + """Yield (raw_line, requirement_part, comment_part) for real requirement lines.""" + for raw_line in path.read_text().splitlines(): + stripped = raw_line.strip() + if not stripped or stripped.startswith("#"): + continue + req_part, _, comment_part = stripped.partition("#") + req_part = req_part.strip() + if not req_part: + continue + yield raw_line, req_part, comment_part.strip() + + +def _find_requirement(path: pathlib.Path, package_name: str): + """Find the requirement line for `package_name` (case-insensitive, ignoring extras).""" + name_re = re.compile(r"^([A-Za-z0-9_.\-]+)") + for raw_line, req_part, comment_part in _iter_non_comment_lines(path): + m = name_re.match(req_part) + if m and m.group(1).lower() == package_name.lower(): + return raw_line, req_part, comment_part + return None + + +# --------------------------------------------------------------------------- +# core/requirements.txt +# --------------------------------------------------------------------------- + + +class TestCoreRequirementsLitellmPin: + def _get(self): + found = _find_requirement(CORE_REQUIREMENTS, "litellm") + assert found is not None, "litellm entry missing from core/requirements.txt" + return found + + def test_litellm_pin_present_and_parseable(self): + _, req_part, _ = self._get() + req = Requirement(req_part) + assert req.name == "litellm" + assert "caching" in req.extras + + def test_litellm_pin_fixes_ghsa_4xpc_pv4p_pm3w(self): + """litellm must be pinned to a version >= 1.84.0 (the fix for the API-key leak GHSA).""" + _, req_part, _ = self._get() + req = Requirement(req_part) + pinned_version = next(iter(req.specifier)).version + assert Version(pinned_version) >= Version("1.84.0") + + def test_litellm_pin_is_not_the_old_vulnerable_version(self): + _, req_part, _ = self._get() + assert "1.83.14" not in req_part + + def test_litellm_comment_documents_the_cve(self): + _, _, comment_part = self._get() + assert "GHSA-4xpc-pv4p-pm3w" in comment_part + + +class TestCoreRequirementsWandbFloor: + def _get(self): + found = _find_requirement(CORE_REQUIREMENTS, "wandb") + assert found is not None, "wandb entry missing from core/requirements.txt" + return found + + def test_wandb_is_a_floor_not_an_exact_pin(self): + _, req_part, _ = self._get() + req = Requirement(req_part) + assert req.name == "wandb" + specs = list(req.specifier) + assert len(specs) == 1 + assert specs[0].operator == ">=" + + def test_wandb_floor_is_at_least_0_27_1(self): + _, req_part, _ = self._get() + req = Requirement(req_part) + assert Version("0.27.1") in req.specifier + assert Version("0.27.0") not in req.specifier + + def test_wandb_comment_documents_the_crypto_fix_and_click_requirement(self): + _, _, comment_part = self._get() + assert "x/crypto" in comment_part + assert "click>=8.2" in comment_part + + +# --------------------------------------------------------------------------- +# requirements/pipeline.txt +# --------------------------------------------------------------------------- + + +class TestPipelineRequirements: + def test_click_upper_bound_pin_was_removed(self): + """The `click < 8.2.0` workaround pin must not be reintroduced. + + It was removed because wandb>=0.27.1 requires click>=8.2, and the + upper-bound pin (originally added for ai-dynamo/dynamo#1039) is no + longer needed once typer>=0.16 is used. + """ + found = _find_requirement(PIPELINE_REQUIREMENTS, "click") + assert found is None, f"unexpected explicit click pin re-appeared: {found}" + + def test_typer_floor_is_at_least_0_16(self): + found = _find_requirement(PIPELINE_REQUIREMENTS, "typer") + assert found is not None, "typer entry missing from requirements/pipeline.txt" + _, req_part, _ = found + req = Requirement(req_part) + assert req.name == "typer" + assert Version("0.16") in req.specifier + assert Version("0.15.0") not in req.specifier + + def test_typer_comment_references_click_compatibility(self): + found = _find_requirement(PIPELINE_REQUIREMENTS, "typer") + assert found is not None + _, _, comment_part = found + assert "click 8.2" in comment_part + assert "wandb>=0.27.1" in comment_part + + def test_pipeline_requirements_lines_are_parseable(self): + """Every requirement line in the file should still be valid PEP 508.""" + for raw_line, req_part, _ in _iter_non_comment_lines(PIPELINE_REQUIREMENTS): + try: + Requirement(req_part) + except InvalidRequirement as exc: + pytest.fail(f"Unparseable requirement line {raw_line!r}: {exc}") + + +# --------------------------------------------------------------------------- +# requirements/stem.txt +# --------------------------------------------------------------------------- + + +class TestStemRequirementsLxmlFloor: + def _get(self): + found = _find_requirement(STEM_REQUIREMENTS, "lxml") + assert found is not None, "lxml entry missing from requirements/stem.txt" + return found + + def test_lxml_floor_is_at_least_6_1_0(self): + _, req_part, _ = self._get() + req = Requirement(req_part) + assert req.name == "lxml" + assert Version("6.1.0") in req.specifier + assert Version("6.0.9") not in req.specifier + + def test_lxml_comment_documents_the_ghsa(self): + _, _, comment_part = self._get() + assert "GHSA-vfmq-68hx-4jfw" in comment_part + + def test_lxml_is_no_longer_unpinned(self): + _, req_part, _ = self._get() + assert req_part != "lxml" + + +# --------------------------------------------------------------------------- +# pyproject.toml +# --------------------------------------------------------------------------- + + +@pytest.fixture(scope="module") +def pyproject_data(): + if tomllib is None: + pytest.skip("no TOML parser (tomllib/tomli) available in this environment") + with open(PYPROJECT_TOML, "rb") as f: + return tomllib.load(f) + + +class TestPyprojectUvOverrides: + def test_override_dependencies_still_contains_httpx_and_urllib3(self, pyproject_data): + overrides = pyproject_data["tool"]["uv"]["override-dependencies"] + parsed = {Requirement(o).name: Requirement(o) for o in overrides} + assert "httpx" in parsed + assert "urllib3" in parsed + + def test_httpx_override_floor_unchanged_at_0_28_1(self, pyproject_data): + overrides = pyproject_data["tool"]["uv"]["override-dependencies"] + httpx_entry = next(o for o in overrides if Requirement(o).name == "httpx") + req = Requirement(httpx_entry) + assert "http2" in req.extras + assert Version("0.28.1") in req.specifier + assert Version("0.27.2") not in req.specifier + + def test_urllib3_override_unchanged(self, pyproject_data): + overrides = pyproject_data["tool"]["uv"]["override-dependencies"] + urllib3_entry = next(o for o in overrides if Requirement(o).name == "urllib3") + req = Requirement(urllib3_entry) + assert Version("2.6.3") in req.specifier + assert Version("1.26.0") not in req.specifier + + def test_override_dependencies_are_all_parseable(self, pyproject_data): + overrides = pyproject_data["tool"]["uv"]["override-dependencies"] + for entry in overrides: + try: + Requirement(entry) + except InvalidRequirement as exc: + pytest.fail(f"Unparseable override-dependencies entry {entry!r}: {exc}") + + +class TestPyprojectCommentUpdated: + """The comment above override-dependencies referenced an exact litellm pin + that has since moved; make sure it was updated rather than left stale.""" + + def test_comment_no_longer_references_stale_litellm_pin(self): + text = PYPROJECT_TOML.read_text() + assert "litellm==1.83.14" not in text + + def test_comment_describes_httpx_floor_requirement(self): + text = PYPROJECT_TOML.read_text() + assert "litellm's httpx>=0.28.0 floor" in text + + +# --------------------------------------------------------------------------- +# Cross-file consistency +# --------------------------------------------------------------------------- + + +def test_wandb_and_typer_click_requirement_comments_are_consistent(): + """wandb's comment says it needs click>=8.2; typer's comment (in the + sibling pipeline.txt file) should agree, since typer>=0.16 is the + mechanism that satisfies that click floor without conflicting pins.""" + _, _, wandb_comment = _find_requirement(CORE_REQUIREMENTS, "wandb") + _, _, typer_comment = _find_requirement(PIPELINE_REQUIREMENTS, "typer") + assert "click>=8.2" in wandb_comment + assert "click 8.2" in typer_comment \ No newline at end of file From a800cd0e4d1bd8ddac51aaf2dca59487df796c35 Mon Sep 17 00:00:00 2001 From: "coderabbitai[bot]" <136622811+coderabbitai[bot]@users.noreply.github.com> Date: Thu, 9 Jul 2026 15:49:36 +0000 Subject: [PATCH 32/46] =?UTF-8?q?=F0=9F=93=9D=20CodeRabbit=20Chat:=20Add?= =?UTF-8?q?=20unit=20tests=20for=20PR=20changes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tests/test_requirements_versions.py | 225 ++++++++++++++++++++++++++++ 1 file changed, 225 insertions(+) create mode 100644 tests/test_requirements_versions.py diff --git a/tests/test_requirements_versions.py b/tests/test_requirements_versions.py new file mode 100644 index 0000000000..38f1cfbdb9 --- /dev/null +++ b/tests/test_requirements_versions.py @@ -0,0 +1,225 @@ +# Copyright (c) 2025, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Regression tests for security-motivated dependency pins. + +This PR bumps several dependency floors/pins to close known CVEs: + * litellm[caching] -> ==1.84.10 (fixes GHSA-4xpc-pv4p-pm3w) + * wandb -> >=0.27.1 (bundled wandb-core Go binary CVEs) + * lxml -> >=6.1.0 (fixes GHSA-vfmq-68hx-4jfw) + * typer -> >=0.16 (click 8.2 compatible) + * click -> pin removed from requirements/pipeline.txt + +It also relies on `[tool.uv].override-dependencies` in pyproject.toml to +relax transitive pins (httpx, urllib3) so a `uv pip`/`uv sync` resolve can +satisfy the new floors. + +These tests parse the actual requirements files and pyproject.toml so that +any future edit which accidentally re-introduces a vulnerable pin (or drops +one of these floors) will fail CI, rather than only being caught during a +manual dependency resolve. +""" + +import re +from pathlib import Path + +import pytest +from packaging.requirements import InvalidRequirement, Requirement +from packaging.version import Version + +REPO_ROOT = Path(__file__).parent.parent + +CORE_REQUIREMENTS = REPO_ROOT / "core" / "requirements.txt" +PIPELINE_REQUIREMENTS = REPO_ROOT / "requirements" / "pipeline.txt" +STEM_REQUIREMENTS = REPO_ROOT / "requirements" / "stem.txt" +PYPROJECT_TOML = REPO_ROOT / "pyproject.toml" + + +def _load_toml(path: Path) -> dict: + try: + import tomllib # Python >= 3.11 + except ImportError: # pragma: no cover - Python 3.10 fallback + tomllib = pytest.importorskip("tomli") + with open(path, "rb") as f: + return tomllib.load(f) + + +def _iter_requirement_lines(path: Path): + """Yield (raw_line, code_part, comment_part) for each non-blank, non-pure-comment line.""" + for raw_line in path.read_text().splitlines(): + stripped = raw_line.strip() + if not stripped or stripped.startswith("#"): + continue + code_part, _, comment_part = raw_line.partition("#") + yield raw_line, code_part.strip(), comment_part.strip() + + +def _find_requirement(path: Path, package_name: str) -> tuple[Requirement, str]: + """Find the requirement line for `package_name` (case-insensitive, ignoring extras). + + Returns a tuple of (parsed Requirement, trailing comment string). + Skips lines that aren't parseable as PEP 508 requirements (e.g. `pkg @ git+...` + URLs, which packaging.Requirement *can* actually parse, but we guard anyway). + """ + for raw_line, code_part, comment_part in _iter_requirement_lines(path): + if not code_part: + continue + try: + req = Requirement(code_part) + except InvalidRequirement: + continue + if req.name.lower() == package_name.lower(): + return req, comment_part + raise AssertionError(f"Could not find requirement '{package_name}' in {path}") + + +class TestCoreRequirements: + """core/requirements.txt: litellm and wandb security floors.""" + + def test_litellm_pin_fixes_ghsa_4xpc_pv4p_pm3w(self): + req, comment = _find_requirement(CORE_REQUIREMENTS, "litellm") + assert "caching" in req.extras, "litellm[caching] extra must be preserved" + + # Must be pinned to an exact version (== specifier) so the resolver is deterministic. + specs = {spec.operator: spec.version for spec in req.specifier} + assert "==" in specs, f"expected an exact pin for litellm, got specifier {req.specifier}" + + pinned_version = Version(specs["=="]) + assert pinned_version >= Version("1.84.0"), ( + f"litellm is pinned to {pinned_version}, which is below the 1.84.0 floor that " + "fixes GHSA-4xpc-pv4p-pm3w (API-key leak to arbitrary Host header)" + ) + assert "GHSA-4xpc-pv4p-pm3w" in comment + + def test_litellm_vulnerable_pin_not_reintroduced(self): + """Regression guard: the old vulnerable exact pin must not come back.""" + content = CORE_REQUIREMENTS.read_text() + assert "litellm[caching]==1.83.14" not in content + assert "1.83.14" not in content + + def test_wandb_pin_fixes_bundled_go_binary_cves(self): + req, comment = _find_requirement(CORE_REQUIREMENTS, "wandb") + specs = {spec.operator: spec.version for spec in req.specifier} + assert ">=" in specs, f"expected a floor (>=) specifier for wandb, got {req.specifier}" + + floor_version = Version(specs[">="]) + assert floor_version >= Version("0.27.1"), ( + f"wandb floor is {floor_version}, which is below 0.27.1 (first release with the " + "patched x/crypto 0.52.0 wandb-core binary)" + ) + assert "click>=8.2" in comment + + def test_wandb_is_not_unbounded_or_unpinned(self): + """wandb must remain a floor-pinned requirement (bare 'wandb' with no version is + the pre-fix state and would allow an unvetted, potentially vulnerable version).""" + for raw_line, code_part, _ in _iter_requirement_lines(CORE_REQUIREMENTS): + if code_part == "wandb": + pytest.fail(f"wandb requirement has no version floor: {raw_line!r}") + + +class TestPipelineRequirements: + """requirements/pipeline.txt: click unpin + typer floor.""" + + def test_click_upper_bound_pin_removed(self): + """The old `click < 8.2.0` pin (needed to work around a typer/click bug) must be gone, + since wandb>=0.27.1 now requires click>=8.2.""" + for raw_line, code_part, _ in _iter_requirement_lines(PIPELINE_REQUIREMENTS): + if not code_part: + continue + try: + req = Requirement(code_part) + except InvalidRequirement: + continue + assert req.name.lower() != "click", ( + f"requirements/pipeline.txt should not pin click directly anymore, found: {raw_line!r}" + ) + + def test_click_pin_line_absent_from_raw_text(self): + """Belt-and-suspenders regression check on the exact removed line.""" + content = PIPELINE_REQUIREMENTS.read_text() + assert "click < 8.2.0" not in content + assert not re.search(r"^click\s*[<>=]", content, re.MULTILINE) + + def test_typer_floor_is_click_8_2_compatible(self): + req, comment = _find_requirement(PIPELINE_REQUIREMENTS, "typer") + specs = {spec.operator: spec.version for spec in req.specifier} + assert ">=" in specs, f"expected a floor (>=) specifier for typer, got {req.specifier}" + + floor_version = Version(specs[">="]) + assert floor_version >= Version("0.16"), ( + f"typer floor is {floor_version}, which is below 0.16 (the first click-8.2-compatible " + "release that also satisfies wandb>=0.27.1's click>=8.2 requirement)" + ) + assert "click 8.2" in comment or "click>=8.2" in comment + + def test_nemo_run_and_launcher_pins_untouched(self): + """Sanity: other pipeline deps referenced by the diff context are still present.""" + _find_requirement(PIPELINE_REQUIREMENTS, "nemo-evaluator-launcher") + content = PIPELINE_REQUIREMENTS.read_text() + assert "nemo_run @ git+https://github.com/NVIDIA-NeMo/Run" in content + + +class TestStemRequirements: + """requirements/stem.txt: lxml security floor.""" + + def test_lxml_pin_fixes_ghsa_vfmq_68hx_4jfw(self): + req, comment = _find_requirement(STEM_REQUIREMENTS, "lxml") + specs = {spec.operator: spec.version for spec in req.specifier} + assert ">=" in specs, f"expected a floor (>=) specifier for lxml, got {req.specifier}" + + floor_version = Version(specs[">="]) + assert floor_version >= Version("6.1.0"), ( + f"lxml floor is {floor_version}, which is below 6.1.0 (fixes GHSA-vfmq-68hx-4jfw)" + ) + assert "GHSA-vfmq-68hx-4jfw" in comment + + def test_lxml_is_not_unbounded_or_unpinned(self): + for raw_line, code_part, _ in _iter_requirement_lines(STEM_REQUIREMENTS): + if code_part == "lxml": + pytest.fail(f"lxml requirement has no version floor: {raw_line!r}") + + +class TestPyprojectUvOverrides: + """pyproject.toml: [tool.uv].override-dependencies still relaxes the transitive pins + that would otherwise conflict with the new litellm floor.""" + + @pytest.fixture(scope="class") + def uv_overrides(self): + data = _load_toml(PYPROJECT_TOML) + overrides = data["tool"]["uv"]["override-dependencies"] + parsed = {} + for entry in overrides: + req = Requirement(entry) + parsed[req.name.lower()] = req + return parsed + + def test_httpx_override_present_for_litellm_compat(self, uv_overrides): + assert "httpx" in uv_overrides, "expected an httpx override in [tool.uv].override-dependencies" + req = uv_overrides["httpx"] + assert "http2" in req.extras + specs = {spec.operator: spec.version for spec in req.specifier} + assert ">=" in specs + assert Version(specs[">="]) >= Version("0.28.1") + + def test_urllib3_override_present(self, uv_overrides): + assert "urllib3" in uv_overrides, "expected a urllib3 override in [tool.uv].override-dependencies" + req = uv_overrides["urllib3"] + specs = {spec.operator: spec.version for spec in req.specifier} + assert ">=" in specs + assert Version(specs[">="]) >= Version("2.6.3") + + def test_dependencies_still_sourced_from_core_and_pipeline_requirements(self): + data = _load_toml(PYPROJECT_TOML) + dynamic_deps = data["tool"]["setuptools"]["dynamic"]["dependencies"] + assert dynamic_deps["file"] == ["core/requirements.txt", "requirements/pipeline.txt"] \ No newline at end of file From 8eb45713eeca6f3654c6fca070ff3e08be66b1b3 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Thu, 9 Jul 2026 11:54:47 -0400 Subject: [PATCH 33/46] test: consolidate generated dependency-pin tests into one suite CodeRabbit's test generation ran twice and committed two near-duplicate suites (test_dependency_pins.py + test_requirements_versions.py) covering the same pins. Keep the more robust one (operator-keyed specifier parsing instead of next(iter(specifier)), which is order-fragile on multi-spec requirements) and graft the three tests unique to the deleted file: pyproject stale-comment guards, pipeline-lines-parseable, and the wandb/typer click-comment consistency check. Also: - fix the copyright year (2026, not 2025) - add tomli (python_version < 3.11) to common-tests.txt so the pyproject override tests actually RUN on the CI's Python 3.10 instead of silently skipping (tomllib is stdlib only from 3.11) - add the missing trailing newline that failed the pre-commit end-of-file-fixer hook on the generated files 17 tests pass on Python 3.10 with the CI's -m 'not gpu' selection; pre-commit (pinned ruff) clean. Signed-off-by: Nick Gupta --- requirements/common-tests.txt | 2 + tests/test_dependency_pins.py | 267 ---------------------------- tests/test_requirements_versions.py | 38 +++- 3 files changed, 38 insertions(+), 269 deletions(-) delete mode 100644 tests/test_dependency_pins.py diff --git a/requirements/common-tests.txt b/requirements/common-tests.txt index 207afd06f3..2771674dea 100644 --- a/requirements/common-tests.txt +++ b/requirements/common-tests.txt @@ -20,3 +20,5 @@ pytest-timeout # ray.job_submission, which lives in the [default] extras. ray[default]>=2.43,<3.0 soundfile +# TOML parsing in tests/test_requirements_versions.py on Python 3.10 (tomllib is stdlib only from 3.11) +tomli; python_version < '3.11' diff --git a/tests/test_dependency_pins.py b/tests/test_dependency_pins.py deleted file mode 100644 index 0986d94eb9..0000000000 --- a/tests/test_dependency_pins.py +++ /dev/null @@ -1,267 +0,0 @@ -# Copyright (c) 2025, NVIDIA CORPORATION. All rights reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -"""Regression tests for security-motivated dependency pins. - -This suite guards the version floors/pins introduced to remediate known -CVEs/GHSAs (litellm host-header key leak, wandb-core x/crypto bundle, lxml -GHSA-vfmq-68hx-4jfw) as well as the click/typer compatibility fix that was -required to unblock the wandb>=0.27.1 upgrade. It does *not* attempt to -validate the full dependency graph -- only the specific lines that were -touched by the security-pin PR, plus enough surrounding structure to catch -someone accidentally reverting/loosening them in the future. -""" - -import pathlib -import re - -import pytest -from packaging.requirements import InvalidRequirement, Requirement -from packaging.version import Version - -try: - import tomllib -except ModuleNotFoundError: # pragma: no cover - py<3.11 fallback - try: - import tomli as tomllib - except ModuleNotFoundError: - tomllib = None - -REPO_ROOT = pathlib.Path(__file__).parent.parent -CORE_REQUIREMENTS = REPO_ROOT / "core" / "requirements.txt" -PIPELINE_REQUIREMENTS = REPO_ROOT / "requirements" / "pipeline.txt" -STEM_REQUIREMENTS = REPO_ROOT / "requirements" / "stem.txt" -PYPROJECT_TOML = REPO_ROOT / "pyproject.toml" - - -def _iter_non_comment_lines(path: pathlib.Path): - """Yield (raw_line, requirement_part, comment_part) for real requirement lines.""" - for raw_line in path.read_text().splitlines(): - stripped = raw_line.strip() - if not stripped or stripped.startswith("#"): - continue - req_part, _, comment_part = stripped.partition("#") - req_part = req_part.strip() - if not req_part: - continue - yield raw_line, req_part, comment_part.strip() - - -def _find_requirement(path: pathlib.Path, package_name: str): - """Find the requirement line for `package_name` (case-insensitive, ignoring extras).""" - name_re = re.compile(r"^([A-Za-z0-9_.\-]+)") - for raw_line, req_part, comment_part in _iter_non_comment_lines(path): - m = name_re.match(req_part) - if m and m.group(1).lower() == package_name.lower(): - return raw_line, req_part, comment_part - return None - - -# --------------------------------------------------------------------------- -# core/requirements.txt -# --------------------------------------------------------------------------- - - -class TestCoreRequirementsLitellmPin: - def _get(self): - found = _find_requirement(CORE_REQUIREMENTS, "litellm") - assert found is not None, "litellm entry missing from core/requirements.txt" - return found - - def test_litellm_pin_present_and_parseable(self): - _, req_part, _ = self._get() - req = Requirement(req_part) - assert req.name == "litellm" - assert "caching" in req.extras - - def test_litellm_pin_fixes_ghsa_4xpc_pv4p_pm3w(self): - """litellm must be pinned to a version >= 1.84.0 (the fix for the API-key leak GHSA).""" - _, req_part, _ = self._get() - req = Requirement(req_part) - pinned_version = next(iter(req.specifier)).version - assert Version(pinned_version) >= Version("1.84.0") - - def test_litellm_pin_is_not_the_old_vulnerable_version(self): - _, req_part, _ = self._get() - assert "1.83.14" not in req_part - - def test_litellm_comment_documents_the_cve(self): - _, _, comment_part = self._get() - assert "GHSA-4xpc-pv4p-pm3w" in comment_part - - -class TestCoreRequirementsWandbFloor: - def _get(self): - found = _find_requirement(CORE_REQUIREMENTS, "wandb") - assert found is not None, "wandb entry missing from core/requirements.txt" - return found - - def test_wandb_is_a_floor_not_an_exact_pin(self): - _, req_part, _ = self._get() - req = Requirement(req_part) - assert req.name == "wandb" - specs = list(req.specifier) - assert len(specs) == 1 - assert specs[0].operator == ">=" - - def test_wandb_floor_is_at_least_0_27_1(self): - _, req_part, _ = self._get() - req = Requirement(req_part) - assert Version("0.27.1") in req.specifier - assert Version("0.27.0") not in req.specifier - - def test_wandb_comment_documents_the_crypto_fix_and_click_requirement(self): - _, _, comment_part = self._get() - assert "x/crypto" in comment_part - assert "click>=8.2" in comment_part - - -# --------------------------------------------------------------------------- -# requirements/pipeline.txt -# --------------------------------------------------------------------------- - - -class TestPipelineRequirements: - def test_click_upper_bound_pin_was_removed(self): - """The `click < 8.2.0` workaround pin must not be reintroduced. - - It was removed because wandb>=0.27.1 requires click>=8.2, and the - upper-bound pin (originally added for ai-dynamo/dynamo#1039) is no - longer needed once typer>=0.16 is used. - """ - found = _find_requirement(PIPELINE_REQUIREMENTS, "click") - assert found is None, f"unexpected explicit click pin re-appeared: {found}" - - def test_typer_floor_is_at_least_0_16(self): - found = _find_requirement(PIPELINE_REQUIREMENTS, "typer") - assert found is not None, "typer entry missing from requirements/pipeline.txt" - _, req_part, _ = found - req = Requirement(req_part) - assert req.name == "typer" - assert Version("0.16") in req.specifier - assert Version("0.15.0") not in req.specifier - - def test_typer_comment_references_click_compatibility(self): - found = _find_requirement(PIPELINE_REQUIREMENTS, "typer") - assert found is not None - _, _, comment_part = found - assert "click 8.2" in comment_part - assert "wandb>=0.27.1" in comment_part - - def test_pipeline_requirements_lines_are_parseable(self): - """Every requirement line in the file should still be valid PEP 508.""" - for raw_line, req_part, _ in _iter_non_comment_lines(PIPELINE_REQUIREMENTS): - try: - Requirement(req_part) - except InvalidRequirement as exc: - pytest.fail(f"Unparseable requirement line {raw_line!r}: {exc}") - - -# --------------------------------------------------------------------------- -# requirements/stem.txt -# --------------------------------------------------------------------------- - - -class TestStemRequirementsLxmlFloor: - def _get(self): - found = _find_requirement(STEM_REQUIREMENTS, "lxml") - assert found is not None, "lxml entry missing from requirements/stem.txt" - return found - - def test_lxml_floor_is_at_least_6_1_0(self): - _, req_part, _ = self._get() - req = Requirement(req_part) - assert req.name == "lxml" - assert Version("6.1.0") in req.specifier - assert Version("6.0.9") not in req.specifier - - def test_lxml_comment_documents_the_ghsa(self): - _, _, comment_part = self._get() - assert "GHSA-vfmq-68hx-4jfw" in comment_part - - def test_lxml_is_no_longer_unpinned(self): - _, req_part, _ = self._get() - assert req_part != "lxml" - - -# --------------------------------------------------------------------------- -# pyproject.toml -# --------------------------------------------------------------------------- - - -@pytest.fixture(scope="module") -def pyproject_data(): - if tomllib is None: - pytest.skip("no TOML parser (tomllib/tomli) available in this environment") - with open(PYPROJECT_TOML, "rb") as f: - return tomllib.load(f) - - -class TestPyprojectUvOverrides: - def test_override_dependencies_still_contains_httpx_and_urllib3(self, pyproject_data): - overrides = pyproject_data["tool"]["uv"]["override-dependencies"] - parsed = {Requirement(o).name: Requirement(o) for o in overrides} - assert "httpx" in parsed - assert "urllib3" in parsed - - def test_httpx_override_floor_unchanged_at_0_28_1(self, pyproject_data): - overrides = pyproject_data["tool"]["uv"]["override-dependencies"] - httpx_entry = next(o for o in overrides if Requirement(o).name == "httpx") - req = Requirement(httpx_entry) - assert "http2" in req.extras - assert Version("0.28.1") in req.specifier - assert Version("0.27.2") not in req.specifier - - def test_urllib3_override_unchanged(self, pyproject_data): - overrides = pyproject_data["tool"]["uv"]["override-dependencies"] - urllib3_entry = next(o for o in overrides if Requirement(o).name == "urllib3") - req = Requirement(urllib3_entry) - assert Version("2.6.3") in req.specifier - assert Version("1.26.0") not in req.specifier - - def test_override_dependencies_are_all_parseable(self, pyproject_data): - overrides = pyproject_data["tool"]["uv"]["override-dependencies"] - for entry in overrides: - try: - Requirement(entry) - except InvalidRequirement as exc: - pytest.fail(f"Unparseable override-dependencies entry {entry!r}: {exc}") - - -class TestPyprojectCommentUpdated: - """The comment above override-dependencies referenced an exact litellm pin - that has since moved; make sure it was updated rather than left stale.""" - - def test_comment_no_longer_references_stale_litellm_pin(self): - text = PYPROJECT_TOML.read_text() - assert "litellm==1.83.14" not in text - - def test_comment_describes_httpx_floor_requirement(self): - text = PYPROJECT_TOML.read_text() - assert "litellm's httpx>=0.28.0 floor" in text - - -# --------------------------------------------------------------------------- -# Cross-file consistency -# --------------------------------------------------------------------------- - - -def test_wandb_and_typer_click_requirement_comments_are_consistent(): - """wandb's comment says it needs click>=8.2; typer's comment (in the - sibling pipeline.txt file) should agree, since typer>=0.16 is the - mechanism that satisfies that click floor without conflicting pins.""" - _, _, wandb_comment = _find_requirement(CORE_REQUIREMENTS, "wandb") - _, _, typer_comment = _find_requirement(PIPELINE_REQUIREMENTS, "typer") - assert "click>=8.2" in wandb_comment - assert "click 8.2" in typer_comment \ No newline at end of file diff --git a/tests/test_requirements_versions.py b/tests/test_requirements_versions.py index 38f1cfbdb9..33becac1eb 100644 --- a/tests/test_requirements_versions.py +++ b/tests/test_requirements_versions.py @@ -1,4 +1,4 @@ -# Copyright (c) 2025, NVIDIA CORPORATION. All rights reserved. +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. @@ -222,4 +222,38 @@ def test_urllib3_override_present(self, uv_overrides): def test_dependencies_still_sourced_from_core_and_pipeline_requirements(self): data = _load_toml(PYPROJECT_TOML) dynamic_deps = data["tool"]["setuptools"]["dynamic"]["dependencies"] - assert dynamic_deps["file"] == ["core/requirements.txt", "requirements/pipeline.txt"] \ No newline at end of file + assert dynamic_deps["file"] == ["core/requirements.txt", "requirements/pipeline.txt"] + + +class TestPyprojectCommentUpdated: + """The comment above override-dependencies referenced an exact litellm pin + that has since moved; make sure it was updated rather than left stale.""" + + def test_comment_no_longer_references_stale_litellm_pin(self): + text = PYPROJECT_TOML.read_text() + assert "litellm==1.83.14" not in text + + def test_comment_describes_httpx_floor_requirement(self): + text = PYPROJECT_TOML.read_text() + assert "litellm's httpx>=0.28.0 floor" in text + + +def test_pipeline_requirements_lines_are_parseable(): + """Every requirement line in the file should still be valid PEP 508.""" + for raw_line, code_part, _ in _iter_requirement_lines(PIPELINE_REQUIREMENTS): + if not code_part: + continue + try: + Requirement(code_part) + except InvalidRequirement as exc: + pytest.fail(f"Unparseable requirement line {raw_line!r}: {exc}") + + +def test_wandb_and_typer_click_requirement_comments_are_consistent(): + """wandb's comment says it needs click>=8.2; typer's comment (in the + sibling pipeline.txt file) should agree, since typer>=0.16 is the + mechanism that satisfies that click floor without conflicting pins.""" + _, wandb_comment = _find_requirement(CORE_REQUIREMENTS, "wandb") + _, typer_comment = _find_requirement(PIPELINE_REQUIREMENTS, "typer") + assert "click>=8.2" in wandb_comment + assert "click 8.2" in typer_comment From 92d1582cebedbbd05f66ebe3d99d8fb2ec05621d Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Mon, 13 Jul 2026 14:51:00 -0400 Subject: [PATCH 34/46] test(security): functional tests that NeMo-Skills works with the bumped deps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit test_requirements_versions.py only asserts the pins statically. Add functional coverage that drives litellm 1.84.10, typer/click, and wandb through NeMo-Skills' own code paths (CPU-only, hermetic — no sandbox, no live endpoint, no API keys) so a resolve to a behavior-divergent version fails CI, not a production run: * litellm: OpenAIModel.litellm_kwargs binds api_key to the configured base_url/api_base only (GHSA-4xpc-pv4p-pm3w regression), generate_async calls litellm.acompletion with those credentials and parses the 1.84 response, and the imported litellm exception/type surface still exists. * typer/click: ns CLI --help + per-command help render Parameter.make_metavar (the click 8.2 break typer>=0.16 fixes) and unknown commands are usage errors. * wandb: log_random_samples matches the wandb 0.28 init/save/summary/finish contract, and a real offline init/finish cycle runs with no account. * lxml: importorskip-guarded real parse (optional stem extra). Signed-off-by: Nick Gupta --- tests/test_dependency_functional.py | 263 ++++++++++++++++++++++++++++ 1 file changed, 263 insertions(+) create mode 100644 tests/test_dependency_functional.py diff --git a/tests/test_dependency_functional.py b/tests/test_dependency_functional.py new file mode 100644 index 0000000000..00dbc36f95 --- /dev/null +++ b/tests/test_dependency_functional.py @@ -0,0 +1,263 @@ +# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Functional tests that NeMo-Skills still works with the versions bumped in this PR. + +`tests/test_requirements_versions.py` is a *static* guard — it parses the +requirements files and asserts the pins stay in place. It never imports or runs +the bumped packages. These tests close that gap: they drive the bumped +dependencies **through NeMo-Skills' own code paths** so a resolve that installs a +version whose behavior diverged from what NeMo-Skills expects fails CI, not a +production run. + +Bumps under test: + * litellm[caching] ==1.84.10 (fixes GHSA-4xpc-pv4p-pm3w — the pre-1.84 client + could leak the configured api_key to an + attacker-controlled Host header) + * wandb >=0.27.1 + * typer >=0.16 / click cap removed (typer<0.16 broke on click 8.2's + Parameter.make_metavar signature — dynamo#1039) + * lxml >=6.1.0 (fixes GHSA-vfmq-68hx-4jfw; optional `stem` extra, + so guarded by importorskip) + +All tests are CPU-only, hermetic (no sandbox container, no live LLM endpoint, no +API keys) so they run in the existing `unit-tests` (`-m "not gpu"`) CI job. +""" + +import asyncio +import json +from types import SimpleNamespace +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +# --------------------------------------------------------------------------- +# litellm 1.84.10 — driven through nemo_skills.inference.model +# --------------------------------------------------------------------------- + +_URL = "http://regression-host:1234/v1" +_KEY = "sk-regression-secret" + + +def _make_openai_model(**overrides): + """Construct a real OpenAIModel offline (no tokenizer, no network I/O). + + Default construction does no HTTP: require_tokenizer defaults False, and an + explicit api_key/base_url skip every env-var lookup. + """ + from nemo_skills.inference.model.openai import OpenAIModel + + kwargs = dict(model="dummy-model", base_url=_URL, api_key=_KEY) + kwargs.update(overrides) + return OpenAIModel(**kwargs) + + +def test_openai_model_registered_and_constructs(): + """get_model('openai') resolves and builds under litellm 1.84.""" + from nemo_skills.inference.model import get_model + from nemo_skills.inference.model.openai import OpenAIModel + + model = get_model(server_type="openai", model="dummy-model", base_url=_URL, api_key=_KEY) + assert isinstance(model, OpenAIModel) + + +def test_credentials_bound_to_configured_base_url(): + """CVE regression (GHSA-4xpc-pv4p-pm3w): the api_key is assembled bound to + our configured base_url/api_base only — never a caller-influenced host.""" + model = _make_openai_model() + assert model.litellm_kwargs["api_key"] == _KEY + assert model.litellm_kwargs["base_url"] == _URL + assert model.litellm_kwargs["api_base"] == _URL + # provider-prefixed model name is what litellm routes on + assert model.litellm_kwargs["model"] == "openai/dummy-model" + + +def test_generate_async_calls_litellm_with_bound_credentials(): + """The real generate_async path builds request params and calls + litellm.acompletion; the key travels only alongside our base_url/api_base, + and litellm 1.84's response object parses back into NeMo-Skills' dict.""" + model = _make_openai_model() + + # litellm returns pydantic objects; mimic the attributes NeMo-Skills reads + # plus model_dump() (used by _serialize_output for conversation history). + fake_choice = SimpleNamespace( + message=SimpleNamespace(content="hello world"), + finish_reason="stop", + logprobs=None, + model_dump=lambda: {"message": {"role": "assistant", "content": "hello world"}}, + ) + fake_response = SimpleNamespace( + choices=[fake_choice], + usage=SimpleNamespace(completion_tokens=2, prompt_tokens=3), + ) + + with patch("litellm.acompletion", new=AsyncMock(return_value=fake_response)) as mock_acompletion: + result = asyncio.run( + model.generate_async( + [{"role": "user", "content": "hi"}], + tokens_to_generate=8, + remove_stop_phrases=False, + ) + ) + + assert result["generation"] == "hello world" + assert result["num_generated_tokens"] == 2 + assert result["num_input_tokens"] == 3 + + mock_acompletion.assert_awaited_once() + call_kwargs = mock_acompletion.await_args.kwargs + assert call_kwargs["api_key"] == _KEY + assert call_kwargs["base_url"] == _URL + assert call_kwargs["api_base"] == _URL + assert call_kwargs["model"] == "openai/dummy-model" + # the user message we passed must be the one litellm receives + assert call_kwargs["messages"] == [{"role": "user", "content": "hi"}] + + +def test_build_chat_request_params_shape(): + """The param names NeMo-Skills sends still match litellm 1.84's chat schema.""" + model = _make_openai_model() + params = model._build_chat_request_params( + messages=[{"role": "user", "content": "hi"}], + tokens_to_generate=16, + temperature=0.0, + top_p=0.95, + top_k=-1, + min_p=0.0, + repetition_penalty=1.0, + random_seed=1234, + stop_phrases=None, + timeout=60, + top_logprobs=None, + stream=False, + reasoning_effort=None, + ) + assert params["messages"] == [{"role": "user", "content": "hi"}] + assert params["max_completion_tokens"] == 16 + assert params["seed"] == 1234 + assert params["stream"] is False + + +def test_litellm_exception_and_type_surface_present(): + """NeMo-Skills imports these litellm symbols at module load; guard the API + surface so a litellm bump that relocated them fails here, not at import of + inference/mcp code.""" + from litellm.exceptions import ContextWindowExceededError # used by model/utils.py + from litellm.types.utils import ChatCompletionMessageToolCall # used by mcp/adapters.py + + assert issubclass(ContextWindowExceededError, Exception) + assert ChatCompletionMessageToolCall is not None + + +# --------------------------------------------------------------------------- +# typer >=0.16 / click cap removed — driven through the `ns` CLI app +# --------------------------------------------------------------------------- + + +@pytest.fixture(scope="module") +def cli_app(): + from nemo_skills.pipeline.cli import app + + return app + + +def test_cli_help_renders(cli_app): + """Top-level --help exercises click 8.2's Parameter.make_metavar — the exact + path typer<0.16 crashed on (dynamo#1039). Must render cleanly under the + uncapped click.""" + from typer.testing import CliRunner + + result = CliRunner().invoke(cli_app, ["--help"]) + assert result.exit_code == 0, result.output + assert "Usage" in result.output + + +@pytest.mark.parametrize("subcommand", ["generate", "eval", "prepare_data"]) +def test_cli_subcommand_help_renders(cli_app, subcommand): + """Per-command help proves each registered command's params render metavars + under click 8.2 (the break was per-parameter, so exercise real commands).""" + from typer.testing import CliRunner + + result = CliRunner().invoke(cli_app, [subcommand, "--help"]) + assert result.exit_code == 0, result.output + + +def test_cli_unknown_command_is_usage_error(cli_app): + """Arg parsing still rejects an unknown command (proves the click parser is + wired, not just that --help short-circuits).""" + from typer.testing import CliRunner + + result = CliRunner().invoke(cli_app, ["definitely-not-a-real-command"]) + assert result.exit_code != 0 + + +# --------------------------------------------------------------------------- +# wandb >=0.27.1 — driven through nemo_skills.inference.log_samples_wandb +# --------------------------------------------------------------------------- + + +def test_wandb_log_random_samples_call_contract(tmp_path): + """NeMo-Skills' wandb usage (init/save/summary/finish) still matches the + wandb 0.28 API. Patched so it is deterministic and offline.""" + from nemo_skills.inference import log_samples_wandb + + jsonl = tmp_path / "samples.jsonl" + jsonl.write_text("\n".join(json.dumps({"problem": f"q{i}", "generation": f"a{i}"}) for i in range(4)) + "\n") + + fake_wandb = MagicMock() + fake_wandb.summary = {} + with patch.object(log_samples_wandb, "wandb", fake_wandb): + log_samples_wandb.log_random_samples(str(jsonl), num_samples=2, project="regr", name="run1") + + fake_wandb.init.assert_called_once() + assert fake_wandb.init.call_args.kwargs["project"] == "regr" + assert fake_wandb.init.call_args.kwargs["name"] == "run1" + fake_wandb.save.assert_called_once() + fake_wandb.finish.assert_called_once() + assert fake_wandb.summary["num_samples"] == 4 + + +def test_wandb_offline_init_runs(tmp_path, monkeypatch): + """Real wandb 0.28 imports and completes an offline init/finish cycle with no + account or network (guards the bundled wandb-core the CVE bump targets).""" + import wandb + + monkeypatch.setenv("WANDB_MODE", "offline") + monkeypatch.setenv("WANDB_SILENT", "true") + monkeypatch.setenv("WANDB_DIR", str(tmp_path)) + run = wandb.init(project="regr", name="offline-run", dir=str(tmp_path)) + try: + run.log({"metric": 1.0}) + finally: + wandb.finish() + assert (tmp_path / "wandb").exists() + + +# --------------------------------------------------------------------------- +# lxml >=6.1.0 — optional `stem` extra (skips when not installed, e.g. the +# core-only unit-tests job); exercises the real parser when present. +# --------------------------------------------------------------------------- + + +def test_lxml_parses_html(): + """lxml 6.1 parses HTML via both its own etree and as the BeautifulSoup + backend NeMo-Skills' `stem` extra provides. Skips cleanly when lxml is not + installed (it is not in the core/dev set).""" + lxml_html = pytest.importorskip("lxml.html") + tree = lxml_html.fromstring("

hello

") + assert tree.find(".//p").text == "hello" + + bs4 = pytest.importorskip("bs4") + soup = bs4.BeautifulSoup("

hi

", "lxml") + assert soup.find("p").text == "hi" From 2e0184a1981db7df99ade11f4aa4e3913c37cf80 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Thu, 30 Jul 2026 11:46:44 -0400 Subject: [PATCH 35/46] fix(security): remediate nemo-skills high findings Signed-off-by: Nick Gupta --- core/requirements.txt | 10 +++- dockerfiles/Dockerfile.nemo-skills | 31 +++++++++- nemo_skills/inference/eval/bfcl.py | 3 +- tests/test_dependency_functional.py | 30 +++++++++- tests/test_requirements_versions.py | 90 +++++++++++++++++++---------- 5 files changed, 127 insertions(+), 37 deletions(-) diff --git a/core/requirements.txt b/core/requirements.txt index ff5ff67111..09359976fa 100644 --- a/core/requirements.txt +++ b/core/requirements.txt @@ -7,12 +7,16 @@ bs4 compute-eval @ git+https://github.com/NVIDIA/compute-eval.git@e01a5d2 contractions datasets +# Fixes eight High findings through CVE-2026-55415. +datamodel-code-generator>=0.64.0 editdistance evalplus @ git+https://github.com/evalplus/evalplus@c91370f faiss-cpu fire flask func-timeout +# Fixes six High GitPython advisories through GHSA-94p4-4cq8-9g67. +GitPython>=3.1.55 gradio httpx huggingface_hub @@ -47,6 +51,6 @@ sympy torchcodec tqdm transformers -# Floors click to >=8.2 (wandb 0.26.x only requires click>=8.0.1, so the resolver -# was settling on click 8.1.8). Taken from #1507. -wandb>=0.27.1 +# 0.28.1 is paired with the patched wandb-core built in Dockerfile.nemo-skills. +# Pinning keeps the Python/core protocol pair deterministic. +wandb==0.28.1 diff --git a/dockerfiles/Dockerfile.nemo-skills b/dockerfiles/Dockerfile.nemo-skills index f3d92c9f17..ad76e87819 100644 --- a/dockerfiles/Dockerfile.nemo-skills +++ b/dockerfiles/Dockerfile.nemo-skills @@ -1,3 +1,28 @@ +# W&B 0.28.1 still bundles Go 1.26.4, grpc-go 1.82.0, and x/text 0.38.0. +# W&B commit e118409 is its focused July 17 dependency update to Go 1.26.5, +# grpc-go 1.82.1, and x/text 0.40.0. Build only wandb-core here so the final +# image gets the fixed binary without retaining the Go toolchain or source. +ARG WANDB_CORE_COMMIT=e1184091520c9b44aa1096fdb27b2f4bf52f26d7 +FROM golang:1.26.5 AS wandb-core-builder +ARG WANDB_CORE_COMMIT +RUN git init /src/wandb && \ + cd /src/wandb && \ + git remote add origin https://github.com/wandb/wandb.git && \ + git sparse-checkout init --cone && \ + git sparse-checkout set core && \ + git fetch --depth 1 origin "${WANDB_CORE_COMMIT}" && \ + git checkout --detach FETCH_HEAD +RUN cd /src/wandb/core && \ + CGO_ENABLED=0 go build \ + -tags "disable_grpc_modules parquet_read_only" \ + -ldflags "-s -w -X main.commit=${WANDB_CORE_COMMIT}" \ + -mod=vendor \ + -o /wandb-core \ + ./cmd/wandb-core && \ + go version -m /wandb-core | grep -F "go1.26.5" && \ + go version -m /wandb-core | grep -E "google\.golang\.org/grpc[[:space:]]+v1\.82\.1([[:space:]]|$)" && \ + go version -m /wandb-core | grep -E "golang\.org/x/text[[:space:]]+v0\.40\.0([[:space:]]|$)" + # using ubuntu instead of debian for easier apptainer installation on arm64 FROM ubuntu:22.04 @@ -74,9 +99,13 @@ COPY core/requirements.txt /opt/NeMo-Skills/core/requirements.txt RUN pip install git+https://github.com/NVIDIA/NeMo-speech-data-processor@29b9b1ec0ceaf3ffa441c1d01297371b3f8e11d2 ARG CACHEBUST=4 # Install via `uv pip` from the project directory so [tool.uv].override-dependencies -# in pyproject.toml (which relaxes leptonai's httpx==0.27.2 pin so litellm 1.83.x +# in pyproject.toml (which relaxes leptonai's httpx==0.27.2 pin so litellm 1.84.x # can be installed) is picked up. Plain pip ignores [tool.uv] and the resolver fails. RUN cd /opt/NeMo-Skills && uv pip install --system --no-cache-dir \ -r core/requirements.txt -r requirements/pipeline.txt +# Replace W&B's vulnerable release binary with the source-compatible patched core +# built and module-verified above. The copy preserves its executable mode. +COPY --from=wandb-core-builder /wandb-core /usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core +RUN /usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core --version # Fix http mismatch between lepton and dggs by manually downloading dggs here RUN pip install ddgs diff --git a/nemo_skills/inference/eval/bfcl.py b/nemo_skills/inference/eval/bfcl.py index 7452aa14af..cff8a9e5fb 100644 --- a/nemo_skills/inference/eval/bfcl.py +++ b/nemo_skills/inference/eval/bfcl.py @@ -66,7 +66,8 @@ "cohere==5.18.0", "typer>=0.12.5", "tabulate>=0.9.0", - "datamodel-code-generator==0.25.7", + # 0.64.0 fixes eight High findings through CVE-2026-55415. + "datamodel-code-generator==0.64.0", "google-genai>=1.52.0", # "qwen-agent", # disabling due to some issues (and shouldn't be needed) "mpmath==1.3.0", diff --git a/tests/test_dependency_functional.py b/tests/test_dependency_functional.py index 00dbc36f95..645f01752f 100644 --- a/tests/test_dependency_functional.py +++ b/tests/test_dependency_functional.py @@ -25,7 +25,9 @@ * litellm[caching] ==1.84.10 (fixes GHSA-4xpc-pv4p-pm3w — the pre-1.84 client could leak the configured api_key to an attacker-controlled Host header) - * wandb >=0.27.1 + * GitPython >=3.1.55 + * datamodel-code-generator >=0.64.0 + * wandb ==0.28.1, with a patched core in the container * typer >=0.16 / click cap removed (typer<0.16 broke on click 8.2's Parameter.make_metavar signature — dynamo#1039) * lxml >=6.1.0 (fixes GHSA-vfmq-68hx-4jfw; optional `stem` extra, @@ -37,11 +39,35 @@ import asyncio import json +from importlib.metadata import version from types import SimpleNamespace from unittest.mock import AsyncMock, MagicMock, patch import pytest +# --------------------------------------------------------------------------- +# GitPython >=3.1.55 and datamodel-code-generator >=0.64.0 +# --------------------------------------------------------------------------- + + +def test_gitpython_can_initialize_and_inspect_repository(tmp_path): + import git + from packaging.version import Version + + assert Version(version("GitPython")) >= Version("3.1.55") + repo = git.Repo.init(tmp_path) + assert not repo.bare + assert repo.git_dir == str(tmp_path / ".git") + + +def test_datamodel_code_generator_imports_at_fixed_version(): + import datamodel_code_generator + from packaging.version import Version + + assert datamodel_code_generator is not None + assert Version(version("datamodel-code-generator")) >= Version("0.64.0") + + # --------------------------------------------------------------------------- # litellm 1.84.10 — driven through nemo_skills.inference.model # --------------------------------------------------------------------------- @@ -203,7 +229,7 @@ def test_cli_unknown_command_is_usage_error(cli_app): # --------------------------------------------------------------------------- -# wandb >=0.27.1 — driven through nemo_skills.inference.log_samples_wandb +# wandb ==0.28.1 — driven through nemo_skills.inference.log_samples_wandb # --------------------------------------------------------------------------- diff --git a/tests/test_requirements_versions.py b/tests/test_requirements_versions.py index 33becac1eb..b5dd25ef5d 100644 --- a/tests/test_requirements_versions.py +++ b/tests/test_requirements_versions.py @@ -16,7 +16,9 @@ This PR bumps several dependency floors/pins to close known CVEs: * litellm[caching] -> ==1.84.10 (fixes GHSA-4xpc-pv4p-pm3w) - * wandb -> >=0.27.1 (bundled wandb-core Go binary CVEs) + * GitPython -> >=3.1.55 (fixes six High findings) + * datamodel-code-generator -> >=0.64.0 (fixes eight High findings) + * wandb -> ==0.28.1, paired with a patched wandb-core * lxml -> >=6.1.0 (fixes GHSA-vfmq-68hx-4jfw) * typer -> >=0.16 (click 8.2 compatible) * click -> pin removed from requirements/pipeline.txt @@ -44,6 +46,8 @@ PIPELINE_REQUIREMENTS = REPO_ROOT / "requirements" / "pipeline.txt" STEM_REQUIREMENTS = REPO_ROOT / "requirements" / "stem.txt" PYPROJECT_TOML = REPO_ROOT / "pyproject.toml" +BFCL_MODULE = REPO_ROOT / "nemo_skills" / "inference" / "eval" / "bfcl.py" +NEMO_SKILLS_DOCKERFILE = REPO_ROOT / "dockerfiles" / "Dockerfile.nemo-skills" def _load_toml(path: Path) -> dict: @@ -85,10 +89,10 @@ def _find_requirement(path: Path, package_name: str) -> tuple[Requirement, str]: class TestCoreRequirements: - """core/requirements.txt: litellm and wandb security floors.""" + """core/requirements.txt security floors and pins.""" def test_litellm_pin_fixes_ghsa_4xpc_pv4p_pm3w(self): - req, comment = _find_requirement(CORE_REQUIREMENTS, "litellm") + req, _ = _find_requirement(CORE_REQUIREMENTS, "litellm") assert "caching" in req.extras, "litellm[caching] extra must be preserved" # Must be pinned to an exact version (== specifier) so the resolver is deterministic. @@ -100,32 +104,68 @@ def test_litellm_pin_fixes_ghsa_4xpc_pv4p_pm3w(self): f"litellm is pinned to {pinned_version}, which is below the 1.84.0 floor that " "fixes GHSA-4xpc-pv4p-pm3w (API-key leak to arbitrary Host header)" ) - assert "GHSA-4xpc-pv4p-pm3w" in comment + assert "GHSA-4xpc-pv4p-pm3w" in CORE_REQUIREMENTS.read_text() def test_litellm_vulnerable_pin_not_reintroduced(self): """Regression guard: the old vulnerable exact pin must not come back.""" content = CORE_REQUIREMENTS.read_text() assert "litellm[caching]==1.83.14" not in content - assert "1.83.14" not in content - def test_wandb_pin_fixes_bundled_go_binary_cves(self): - req, comment = _find_requirement(CORE_REQUIREMENTS, "wandb") + def test_gitpython_floor_fixes_all_six_high_findings(self): + req, _ = _find_requirement(CORE_REQUIREMENTS, "GitPython") specs = {spec.operator: spec.version for spec in req.specifier} - assert ">=" in specs, f"expected a floor (>=) specifier for wandb, got {req.specifier}" + assert ">=" in specs, f"expected a floor (>=) specifier for GitPython, got {req.specifier}" + assert Version(specs[">="]) >= Version("3.1.55") - floor_version = Version(specs[">="]) - assert floor_version >= Version("0.27.1"), ( - f"wandb floor is {floor_version}, which is below 0.27.1 (first release with the " - "patched x/crypto 0.52.0 wandb-core binary)" - ) - assert "click>=8.2" in comment + def test_datamodel_code_generator_floor_fixes_all_eight_high_findings(self): + req, _ = _find_requirement(CORE_REQUIREMENTS, "datamodel-code-generator") + specs = {spec.operator: spec.version for spec in req.specifier} + assert ">=" in specs, f"expected a floor (>=) specifier for datamodel-code-generator, got {req.specifier}" + assert Version(specs[">="]) >= Version("0.64.0") + + def test_bfcl_does_not_reinstall_vulnerable_datamodel_code_generator(self): + content = BFCL_MODULE.read_text() + assert '"datamodel-code-generator==0.64.0"' in content + assert "datamodel-code-generator==0.25.7" not in content + + def test_wandb_python_version_is_pinned_to_patched_core_pair(self): + req, _ = _find_requirement(CORE_REQUIREMENTS, "wandb") + specs = {spec.operator: spec.version for spec in req.specifier} + assert specs == {"==": "0.28.1"} def test_wandb_is_not_unbounded_or_unpinned(self): - """wandb must remain a floor-pinned requirement (bare 'wandb' with no version is - the pre-fix state and would allow an unvetted, potentially vulnerable version).""" + """A bare wandb requirement could resolve to a release with a vulnerable core.""" for raw_line, code_part, _ in _iter_requirement_lines(CORE_REQUIREMENTS): if code_part == "wandb": - pytest.fail(f"wandb requirement has no version floor: {raw_line!r}") + pytest.fail(f"wandb requirement has no version pin: {raw_line!r}") + + +class TestPatchedWandbCoreDockerBuild: + """The final image must replace and verify W&B's bundled Go executable.""" + + @pytest.fixture(scope="class") + def dockerfile(self): + return NEMO_SKILLS_DOCKERFILE.read_text() + + def test_immutable_upstream_security_commit_is_pinned(self, dockerfile): + assert "WANDB_CORE_COMMIT=e1184091520c9b44aa1096fdb27b2f4bf52f26d7" in dockerfile + + @pytest.mark.parametrize( + "expected", + [ + "FROM golang:1.26.5 AS wandb-core-builder", + 'go version -m /wandb-core | grep -F "go1.26.5"', + "google\\.golang\\.org/grpc[[:space:]]+v1\\.82\\.1", + "golang\\.org/x/text[[:space:]]+v0\\.40\\.0", + ], + ) + def test_fixed_go_components_are_build_time_verified(self, dockerfile, expected): + assert expected in dockerfile + + def test_verified_binary_replaces_wandb_release_binary(self, dockerfile): + destination = "/usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core" + assert f"COPY --from=wandb-core-builder /wandb-core {destination}" in dockerfile + assert f"RUN {destination} --version" in dockerfile class TestPipelineRequirements: @@ -152,16 +192,16 @@ def test_click_pin_line_absent_from_raw_text(self): assert not re.search(r"^click\s*[<>=]", content, re.MULTILINE) def test_typer_floor_is_click_8_2_compatible(self): - req, comment = _find_requirement(PIPELINE_REQUIREMENTS, "typer") + req, _ = _find_requirement(PIPELINE_REQUIREMENTS, "typer") specs = {spec.operator: spec.version for spec in req.specifier} assert ">=" in specs, f"expected a floor (>=) specifier for typer, got {req.specifier}" floor_version = Version(specs[">="]) assert floor_version >= Version("0.16"), ( f"typer floor is {floor_version}, which is below 0.16 (the first click-8.2-compatible " - "release that also satisfies wandb>=0.27.1's click>=8.2 requirement)" + "release for click 8.2)" ) - assert "click 8.2" in comment or "click>=8.2" in comment + assert "click 8.2" in PIPELINE_REQUIREMENTS.read_text() def test_nemo_run_and_launcher_pins_untouched(self): """Sanity: other pipeline deps referenced by the diff context are still present.""" @@ -247,13 +287,3 @@ def test_pipeline_requirements_lines_are_parseable(): Requirement(code_part) except InvalidRequirement as exc: pytest.fail(f"Unparseable requirement line {raw_line!r}: {exc}") - - -def test_wandb_and_typer_click_requirement_comments_are_consistent(): - """wandb's comment says it needs click>=8.2; typer's comment (in the - sibling pipeline.txt file) should agree, since typer>=0.16 is the - mechanism that satisfies that click floor without conflicting pins.""" - _, wandb_comment = _find_requirement(CORE_REQUIREMENTS, "wandb") - _, typer_comment = _find_requirement(PIPELINE_REQUIREMENTS, "typer") - assert "click>=8.2" in wandb_comment - assert "click 8.2" in typer_comment From b866cf057860064ef2d417b664e481973c1abe05 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Thu, 30 Jul 2026 11:50:18 -0400 Subject: [PATCH 36/46] chore: satisfy requirements sorting hook Signed-off-by: Nick Gupta --- core/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/core/requirements.txt b/core/requirements.txt index 09359976fa..1c707e6e86 100644 --- a/core/requirements.txt +++ b/core/requirements.txt @@ -6,9 +6,9 @@ bs4 compute-eval @ git+https://github.com/NVIDIA/compute-eval.git@e01a5d2 contractions -datasets # Fixes eight High findings through CVE-2026-55415. datamodel-code-generator>=0.64.0 +datasets editdistance evalplus @ git+https://github.com/evalplus/evalplus@c91370f faiss-cpu From bfee7357b6f660425adddb0802d08745803d51bd Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Thu, 30 Jul 2026 11:55:42 -0400 Subject: [PATCH 37/46] fix(container): validate patched wandb core help output Signed-off-by: Nick Gupta --- dockerfiles/Dockerfile.nemo-skills | 4 +++- tests/test_requirements_versions.py | 3 ++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/dockerfiles/Dockerfile.nemo-skills b/dockerfiles/Dockerfile.nemo-skills index ad76e87819..9468b3f9de 100644 --- a/dockerfiles/Dockerfile.nemo-skills +++ b/dockerfiles/Dockerfile.nemo-skills @@ -25,6 +25,7 @@ RUN cd /src/wandb/core && \ # using ubuntu instead of debian for easier apptainer installation on arm64 FROM ubuntu:22.04 +ARG WANDB_CORE_COMMIT # Install Python and other dependencies RUN apt-get update && \ @@ -106,6 +107,7 @@ RUN cd /opt/NeMo-Skills && uv pip install --system --no-cache-dir \ # Replace W&B's vulnerable release binary with the source-compatible patched core # built and module-verified above. The copy preserves its executable mode. COPY --from=wandb-core-builder /wandb-core /usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core -RUN /usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core --version +RUN /usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core --help 2>&1 | \ + grep -F "Commit SHA: ${WANDB_CORE_COMMIT}" # Fix http mismatch between lepton and dggs by manually downloading dggs here RUN pip install ddgs diff --git a/tests/test_requirements_versions.py b/tests/test_requirements_versions.py index b5dd25ef5d..faf0bb9743 100644 --- a/tests/test_requirements_versions.py +++ b/tests/test_requirements_versions.py @@ -165,7 +165,8 @@ def test_fixed_go_components_are_build_time_verified(self, dockerfile, expected) def test_verified_binary_replaces_wandb_release_binary(self, dockerfile): destination = "/usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core" assert f"COPY --from=wandb-core-builder /wandb-core {destination}" in dockerfile - assert f"RUN {destination} --version" in dockerfile + assert f"RUN {destination} --help 2>&1" in dockerfile + assert 'grep -F "Commit SHA: ${WANDB_CORE_COMMIT}"' in dockerfile class TestPipelineRequirements: From faebf367707377b4e51a3dd267e1fb8a7f3011aa Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Thu, 30 Jul 2026 16:24:27 -0400 Subject: [PATCH 38/46] fix(container): remove uv Git cache from runtime image Signed-off-by: Nick Gupta --- dockerfiles/Dockerfile.nemo-skills | 4 ++++ tests/test_requirements_versions.py | 3 +++ 2 files changed, 7 insertions(+) diff --git a/dockerfiles/Dockerfile.nemo-skills b/dockerfiles/Dockerfile.nemo-skills index 9468b3f9de..ed1eee10ec 100644 --- a/dockerfiles/Dockerfile.nemo-skills +++ b/dockerfiles/Dockerfile.nemo-skills @@ -111,3 +111,7 @@ RUN /usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core --help 2>&1 | \ grep -F "Commit SHA: ${WANDB_CORE_COMMIT}" # Fix http mismatch between lepton and dggs by manually downloading dggs here RUN pip install ddgs + +# nSpect's global policy flags Git metadata left by uv's source-distribution +# cache. The cache is build-only, so remove it from the published image. +RUN rm -rf /root/.cache/uv diff --git a/tests/test_requirements_versions.py b/tests/test_requirements_versions.py index faf0bb9743..df76eaf79c 100644 --- a/tests/test_requirements_versions.py +++ b/tests/test_requirements_versions.py @@ -168,6 +168,9 @@ def test_verified_binary_replaces_wandb_release_binary(self, dockerfile): assert f"RUN {destination} --help 2>&1" in dockerfile assert 'grep -F "Commit SHA: ${WANDB_CORE_COMMIT}"' in dockerfile + def test_uv_git_cache_is_removed_from_final_image(self, dockerfile): + assert "RUN rm -rf /root/.cache/uv" in dockerfile + class TestPipelineRequirements: """requirements/pipeline.txt: click unpin + typer floor.""" From bcfcfaf467f39408b4c1df25fa2d0e203a06bd47 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Fri, 31 Jul 2026 10:26:13 -0400 Subject: [PATCH 39/46] fix(security): floor msgpack and setuptools Signed-off-by: Nick Gupta --- core/pyproject.toml | 2 +- dockerfiles/Dockerfile.nemo-skills | 9 ++++++++- pyproject.toml | 7 ++++++- tests/test_dependency_functional.py | 9 +++++++++ tests/test_requirements_versions.py | 28 ++++++++++++++++++++++++++++ tools/pyproject.toml | 2 +- 6 files changed, 53 insertions(+), 4 deletions(-) diff --git a/core/pyproject.toml b/core/pyproject.toml index c5ccb3bd1f..0cc768c30d 100644 --- a/core/pyproject.toml +++ b/core/pyproject.toml @@ -14,7 +14,7 @@ [build-system] requires = [ - "setuptools", + "setuptools>=78.1.1", "wheel" ] build-backend = "setuptools.build_meta" diff --git a/dockerfiles/Dockerfile.nemo-skills b/dockerfiles/Dockerfile.nemo-skills index ed1eee10ec..3cca449a08 100644 --- a/dockerfiles/Dockerfile.nemo-skills +++ b/dockerfiles/Dockerfile.nemo-skills @@ -40,7 +40,7 @@ RUN apt-get update && \ ln -s /usr/bin/python3 /usr/bin/python && \ rm -rf /var/cache/apt/archives /var/lib/apt/lists/* -RUN pip install --upgrade pip setuptools "uv>=0.11.10" +RUN pip install --upgrade pip "setuptools>=78.1.1" "uv>=0.11.10" # Update package lists and install apptainer for arm64 # https://apptainer.org/docs/admin/1.1/installation.html @@ -112,6 +112,13 @@ RUN /usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core --help 2>&1 | \ # Fix http mismatch between lepton and dggs by manually downloading dggs here RUN pip install ddgs +# Guard the final resolved environment against the two High findings seen in +# the July 31 multi-architecture image scan. +RUN python -c "from importlib.metadata import version as v; from packaging.version import Version as V; \ + assert V(v('msgpack')) >= V('1.2.1'), v('msgpack'); \ + assert V(v('setuptools')) >= V('78.1.1'), v('setuptools'); \ + print('msgpack/setuptools security floors OK')" + # nSpect's global policy flags Git metadata left by uv's source-distribution # cache. The cache is build-only, so remove it from the published image. RUN rm -rf /root/.cache/uv diff --git a/pyproject.toml b/pyproject.toml index 617e40e6e9..a3423865ae 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -14,7 +14,7 @@ [build-system] requires = [ - "setuptools", + "setuptools>=78.1.1", "wheel" ] build-backend = "setuptools.build_meta" @@ -80,6 +80,11 @@ override-dependencies = [ # 0.7.0 (a transitive dep of nemo_run) pins urllib3<1.27, but in practice # urllib3>=2 works at runtime, so override the constraint. "urllib3>=2.6.3", + # Container scans found msgpack 1.1.2 (GHSA-6v7p-g79w-8964) and setuptools + # 70.3.0 (CVE-2025-47273) after transitive resolution. Keep the complete + # environment above the first fixed releases. + "msgpack>=1.2.1", + "setuptools>=78.1.1", ] [tool.pytest.ini_options] diff --git a/tests/test_dependency_functional.py b/tests/test_dependency_functional.py index 645f01752f..299db4e97c 100644 --- a/tests/test_dependency_functional.py +++ b/tests/test_dependency_functional.py @@ -32,6 +32,8 @@ Parameter.make_metavar signature — dynamo#1039) * lxml >=6.1.0 (fixes GHSA-vfmq-68hx-4jfw; optional `stem` extra, so guarded by importorskip) + * msgpack >=1.2.1 (fixes GHSA-6v7p-g79w-8964) + * setuptools >=78.1.1 (fixes CVE-2025-47273) All tests are CPU-only, hermetic (no sandbox container, no live LLM endpoint, no API keys) so they run in the existing `unit-tests` (`-m "not gpu"`) CI job. @@ -68,6 +70,13 @@ def test_datamodel_code_generator_imports_at_fixed_version(): assert Version(version("datamodel-code-generator")) >= Version("0.64.0") +def test_msgpack_and_setuptools_security_floors(): + from packaging.version import Version + + assert Version(version("msgpack")) >= Version("1.2.1") + assert Version(version("setuptools")) >= Version("78.1.1") + + # --------------------------------------------------------------------------- # litellm 1.84.10 — driven through nemo_skills.inference.model # --------------------------------------------------------------------------- diff --git a/tests/test_requirements_versions.py b/tests/test_requirements_versions.py index df76eaf79c..64197c8385 100644 --- a/tests/test_requirements_versions.py +++ b/tests/test_requirements_versions.py @@ -20,6 +20,8 @@ * datamodel-code-generator -> >=0.64.0 (fixes eight High findings) * wandb -> ==0.28.1, paired with a patched wandb-core * lxml -> >=6.1.0 (fixes GHSA-vfmq-68hx-4jfw) + * msgpack -> >=1.2.1 (fixes GHSA-6v7p-g79w-8964) + * setuptools -> >=78.1.1 (fixes CVE-2025-47273) * typer -> >=0.16 (click 8.2 compatible) * click -> pin removed from requirements/pipeline.txt @@ -48,6 +50,7 @@ PYPROJECT_TOML = REPO_ROOT / "pyproject.toml" BFCL_MODULE = REPO_ROOT / "nemo_skills" / "inference" / "eval" / "bfcl.py" NEMO_SKILLS_DOCKERFILE = REPO_ROOT / "dockerfiles" / "Dockerfile.nemo-skills" +BUILD_PYPROJECTS = [PYPROJECT_TOML, REPO_ROOT / "core" / "pyproject.toml", REPO_ROOT / "tools" / "pyproject.toml"] def _load_toml(path: Path) -> dict: @@ -171,6 +174,10 @@ def test_verified_binary_replaces_wandb_release_binary(self, dockerfile): def test_uv_git_cache_is_removed_from_final_image(self, dockerfile): assert "RUN rm -rf /root/.cache/uv" in dockerfile + def test_final_image_asserts_msgpack_and_setuptools_floors(self, dockerfile): + assert "V(v('msgpack')) >= V('1.2.1')" in dockerfile + assert "V(v('setuptools')) >= V('78.1.1')" in dockerfile + class TestPipelineRequirements: """requirements/pipeline.txt: click unpin + typer floor.""" @@ -263,6 +270,16 @@ def test_urllib3_override_present(self, uv_overrides): assert ">=" in specs assert Version(specs[">="]) >= Version("2.6.3") + @pytest.mark.parametrize( + ("package", "minimum"), + [("msgpack", "1.2.1"), ("setuptools", "78.1.1")], + ) + def test_container_security_override_present(self, uv_overrides, package, minimum): + assert package in uv_overrides + specs = {spec.operator: spec.version for spec in uv_overrides[package].specifier} + assert ">=" in specs + assert Version(specs[">="]) >= Version(minimum) + def test_dependencies_still_sourced_from_core_and_pipeline_requirements(self): data = _load_toml(PYPROJECT_TOML) dynamic_deps = data["tool"]["setuptools"]["dynamic"]["dependencies"] @@ -291,3 +308,14 @@ def test_pipeline_requirements_lines_are_parseable(): Requirement(code_part) except InvalidRequirement as exc: pytest.fail(f"Unparseable requirement line {raw_line!r}: {exc}") + + +@pytest.mark.parametrize("pyproject", BUILD_PYPROJECTS) +def test_setuptools_build_floor_fixes_cve_2025_47273(pyproject): + data = _load_toml(pyproject) + requirement = next( + Requirement(entry) for entry in data["build-system"]["requires"] if Requirement(entry).name == "setuptools" + ) + specs = {spec.operator: spec.version for spec in requirement.specifier} + assert ">=" in specs + assert Version(specs[">="]) >= Version("78.1.1") diff --git a/tools/pyproject.toml b/tools/pyproject.toml index 4e9c73ee71..1cda69d7ca 100644 --- a/tools/pyproject.toml +++ b/tools/pyproject.toml @@ -14,7 +14,7 @@ [build-system] requires = [ - "setuptools", + "setuptools>=78.1.1", "wheel" ] build-backend = "setuptools.build_meta" From dcf41f65d4e4278b1efa5a729e5c9421b3e50dc5 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Wed, 5 Aug 2026 18:04:47 -0400 Subject: [PATCH 40/46] fix(ray): deliver baked source with working directory Signed-off-by: Nick Gupta --- docs/basics/code-packaging.md | 20 +++++ nemo_skills/pipeline/utils/backends.py | 5 ++ nemo_skills/pipeline/utils/ray_backend.py | 24 ++++- tests/test_backends.py | 103 +++++++++++++++++++++- 4 files changed, 148 insertions(+), 4 deletions(-) diff --git a/docs/basics/code-packaging.md b/docs/basics/code-packaging.md index 1606ddfa04..b8053de367 100644 --- a/docs/basics/code-packaging.md +++ b/docs/basics/code-packaging.md @@ -68,3 +68,23 @@ If you want to have more fine-grained control over code reuse, you can directly While our job submission is somewhat complicated and goes through NeMo-Run, at the end, we simply execute a particular sbatch file that is uploaded to the cluster. It is helpful sometimes to see what's in it and modify directly. You can find sbatch file(s) for each job inside `ssh_tunnel.job_dir` cluster folder that is defined in your cluster config. + +## Ray Jobs code delivery + +The Ray Jobs backend can optionally deliver source code with Ray's native +`runtime_env.working_dir` packaging: + +```yaml +executor: none +backend: + name: ray + dashboard_url: http://:8265 + working_dir: /opt/my-project # local directory or local .zip on the submitter +``` + +When `working_dir` is set, Ray uploads that directory or archive and makes it +the submitted job's current directory. It does not run `pip`, `conda`, or `uv`, +so every dependency must already be present in the Ray worker image. For a +strict-airgap launch, point it at an immutable source directory or archive baked +into the launcher image. If the option is absent, Ray Jobs retain their existing +behavior and no working directory is delivered. diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index 3af8d27e4a..95dfa5d2a8 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -274,6 +274,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa selector = backend_config.get("entrypoint_label_selector") or k8s_cfg.get("entrypoint_label_selector") image_label_key = backend_config.get("image_label_key") or k8s_cfg.get("image_label_key") image_label_selectors = backend_config.get("image_label_selectors") or k8s_cfg.get("image_label_selectors") + working_dir = backend_config.get("working_dir") or k8s_cfg.get("working_dir") image_label_selectors = _resolve_selector_keys_with_container_map( image_label_selectors, cluster_config.get("containers") ) @@ -290,6 +291,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa image_label_key=image_label_key, image_label_selectors=image_label_selectors, env_vars=get_env_variables(cluster_config), + working_dir=working_dir, ) return RayBackend( endpoint=endpoint, @@ -300,6 +302,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa image_label_key=image_label_key, image_label_selectors=image_label_selectors, env_vars=get_env_variables(cluster_config), + working_dir=working_dir, ) if backend_name in {"kubernetes-ray", "ray-kubernetes", "ray_kubernetes"}: k8s_cfg = backend_config.get("kubernetes") or {} @@ -307,6 +310,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa selector = backend_config.get("entrypoint_label_selector") or k8s_cfg.get("entrypoint_label_selector") image_label_key = backend_config.get("image_label_key") or k8s_cfg.get("image_label_key") image_label_selectors = backend_config.get("image_label_selectors") or k8s_cfg.get("image_label_selectors") + working_dir = backend_config.get("working_dir") or k8s_cfg.get("working_dir") image_label_selectors = _resolve_selector_keys_with_container_map( image_label_selectors, cluster_config.get("containers") ) @@ -333,6 +337,7 @@ def get_execution_backend(cluster_config: Dict[str, Any], *, with_ray: bool = Fa image_label_key=image_label_key, image_label_selectors=image_label_selectors, env_vars=get_env_variables(cluster_config), + working_dir=working_dir, ) raise ValueError( diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 1a8aeabb1f..76922fb3fc 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -93,6 +93,7 @@ def __init__( image_label_key: str | None = None, image_label_selectors: Dict[str, Dict[str, str]] | None = None, env_vars: Dict[str, str] | None = None, + working_dir: str | None = None, ): """Configure connection, placement, and env-forwarding options for the backend.""" self.endpoint = endpoint.strip() if endpoint else None @@ -107,6 +108,13 @@ def __init__( # Forwarded to each job's runtime_env so cluster env vars (API keys, # HF_TOKEN, ...) reach the job regardless of the cluster head's env. self.env_vars = {str(k): str(v) for k, v in (env_vars or {}).items() if v is not None} + if working_dir is not None and not isinstance(working_dir, str): + raise ValueError("working_dir must be a string when provided") + # Ray Jobs packages a local directory (or local .zip archive) and makes + # it the job's working directory. This is intentionally the only + # configurable runtime-env code-delivery field: dependency-installing + # fields such as pip and conda are not accepted by this backend. + self.working_dir = (working_dir or "").strip() or None self.kubernetes_mode = None if self.control_plane == "kubernetes": normalized_mode = (kubernetes_mode or "offline").strip().lower() @@ -200,8 +208,15 @@ def _get_jobs_client(self): ) from exc return JobSubmissionClient(self.dashboard_url) - def _build_runtime_env(self) -> Dict[str, Any] | None: - """runtime_env that forwards cluster env vars (API keys, HF_TOKEN, ...) to a job. + def _build_runtime_env(self) -> Dict[str, Any]: + """Build the install-free runtime env for a Ray Job. + + ``working_dir`` is an opt-in code-delivery mechanism. Ray uploads the + configured local directory or local .zip archive through the Jobs API + and makes it the entrypoint's current directory; it does not install a + Python package or resolve dependencies. This lets an airgapped launcher + deliver source already baked into its image without a host checkout, + ``PYTHONPATH`` overlay, or runtime pip/uv operation. Always sets RAY_OVERRIDE_JOB_RUNTIME_ENV=1. A job whose driver calls ray.init() again with its own runtime_env (e.g. NeMo-RL GRPO/rollout) would otherwise hit @@ -212,7 +227,10 @@ def _build_runtime_env(self) -> Dict[str, Any] | None: """ env_vars = dict(self.env_vars) env_vars["RAY_OVERRIDE_JOB_RUNTIME_ENV"] = "1" - return {"env_vars": env_vars} + runtime_env: Dict[str, Any] = {"env_vars": env_vars} + if self.working_dir: + runtime_env["working_dir"] = self.working_dir + return runtime_env def _stop_jobs_best_effort( self, diff --git a/tests/test_backends.py b/tests/test_backends.py index a9f3fbff09..3d4e030e83 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -242,7 +242,108 @@ def test_ray_backend_runtime_env_sets_override_even_with_no_env(): backend = RayBackend(dashboard_url="http://ray-head:8265", env_vars={}) # Even with no forwarded env vars the runtime_env still carries the override flag (never None). - assert backend._build_runtime_env() == {"env_vars": {"RAY_OVERRIDE_JOB_RUNTIME_ENV": "1"}} + runtime_env = backend._build_runtime_env() + assert runtime_env == {"env_vars": {"RAY_OVERRIDE_JOB_RUNTIME_ENV": "1"}} + assert {"working_dir", "py_modules", "pip", "conda"}.isdisjoint(runtime_env) + + +def test_ray_backend_working_dir_enables_install_free_code_delivery(): + """The opt-in working dir is the only runtime-env code-delivery field. + + Ray packages this already-present directory/archive and changes the job cwd; + no dependency installer configuration is synthesized. + """ + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + backend = RayBackend( + dashboard_url="http://ray-head:8265", + env_vars={"AIRGAP": "1"}, + working_dir=" /opt/nvflow ", + ) + + assert backend._build_runtime_env() == { + "env_vars": {"AIRGAP": "1", "RAY_OVERRIDE_JOB_RUNTIME_ENV": "1"}, + "working_dir": "/opt/nvflow", + } + + +def test_ray_job_submission_receives_configured_working_dir(): + from types import SimpleNamespace + + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + submitted = [] + + class RecordingClient: + def submit_job(self, **kwargs): + submitted.append(kwargs) + return "job-1" + + def get_job_status(self, job_id): + assert job_id == "job-1" + return "SUCCEEDED" + + def get_job_logs(self, job_id): + assert job_id == "job-1" + return "" + + backend = RayBackend( + dashboard_url="http://ray-head:8265", + working_dir="/opt/nvflow", + ) + exp = SimpleNamespace(_title="airgap-code-delivery") + + backend._submit_jobs_concurrently( + RecordingClient(), + exp, + [{"task_name": "prepare-data", "command": "python -m nvflow.prepare_data"}], + ) + + assert len(submitted) == 1 + assert submitted[0]["runtime_env"] == { + "env_vars": {"RAY_OVERRIDE_JOB_RUNTIME_ENV": "1"}, + "working_dir": "/opt/nvflow", + } + assert "pip" not in submitted[0]["runtime_env"] + assert "conda" not in submitted[0]["runtime_env"] + + +def test_ray_backend_rejects_non_string_working_dir(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + with pytest.raises(ValueError, match="working_dir must be a string"): + RayBackend(dashboard_url="http://ray-head:8265", working_dir=["/opt/nvflow"]) + + +@pytest.mark.parametrize("backend_name", ["ray", "kubernetes-ray"]) +def test_ray_backend_resolves_working_dir_from_cluster_config(backend_name): + cluster_config = { + "executor": "none", + "backend": { + "name": backend_name, + "dashboard_url": "http://ray-head:8265", + "working_dir": "/opt/nvflow-source.zip", + }, + } + + backend = get_execution_backend(cluster_config) + + assert backend.working_dir == "/opt/nvflow-source.zip" + assert backend._build_runtime_env()["working_dir"] == "/opt/nvflow-source.zip" + + +def test_default_slurm_backend_ignores_ray_working_dir_and_stays_lazy(): + """Ray-only code delivery must not alter the default Slurm backend.""" + backend = get_execution_backend( + { + "executor": "slurm", + "backend": {"name": "default", "working_dir": "/opt/nvflow"}, + } + ) + + assert backend.name == "default" + assert backend.stage_metadata() is None + assert not hasattr(backend, "working_dir") # --------------------------------------------------------------------------- From 982bc66d6232bd2f220d75455d43794c0a27ae0f Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Wed, 5 Aug 2026 18:29:46 -0400 Subject: [PATCH 41/46] fix(ray): preserve delivered code roots after cd Signed-off-by: Nick Gupta --- docs/basics/code-packaging.md | 16 +++++ nemo_skills/pipeline/utils/backends.py | 79 +++++++++++++++++++++++ nemo_skills/pipeline/utils/declarative.py | 35 +++++++--- nemo_skills/pipeline/utils/exp.py | 24 +++++-- nemo_skills/pipeline/utils/ray_backend.py | 43 ++++++++++-- tests/test_backends.py | 55 ++++++++++++++++ tests/test_declarative_pipeline.py | 60 +++++++++++++++++ tests/test_pipeline_utils.py | 79 +++++++++++++++++++++++ 8 files changed, 371 insertions(+), 20 deletions(-) diff --git a/docs/basics/code-packaging.md b/docs/basics/code-packaging.md index b8053de367..63c96e52bc 100644 --- a/docs/basics/code-packaging.md +++ b/docs/basics/code-packaging.md @@ -88,3 +88,19 @@ so every dependency must already be present in the Ray worker image. For a strict-airgap launch, point it at an immutable source directory or archive baked into the launcher image. If the option is absent, Ray Jobs retain their existing behavior and no working directory is delivered. + +Before starting the command, the backend captures Ray's initial uploaded-code +directory in `NEMO_RUN_CODE_DIR`. Legacy `/nemo_run/code/...` command paths are +rewritten to that absolute root, so they continue to work if a workload later +changes directory (for example, `cd /opt/Gym`). Legacy +`/nemo_run/code/nemo_skills/...` paths resolve separately from the NeMo-Skills +package baked into the worker image; the working-directory archive does not need +to duplicate that package. These rewrites apply only to Ray Jobs with an explicit +`working_dir`; local `executor: none`, embedded Ray-on-Slurm, and ordinary Slurm +execution keep their existing path and packaging behavior. + +The backend rewrites paths in generated entrypoint command strings. It does not +edit YAML, JSON, or other files inside the uploaded archive. Configuration files +that must refer to the delivered source after changing directory should resolve +`NEMO_RUN_CODE_DIR` themselves (and may retain `/nemo_run/code` as their non-Ray +default). diff --git a/nemo_skills/pipeline/utils/backends.py b/nemo_skills/pipeline/utils/backends.py index 95dfa5d2a8..256fbb0c4d 100644 --- a/nemo_skills/pipeline/utils/backends.py +++ b/nemo_skills/pipeline/utils/backends.py @@ -45,6 +45,82 @@ def _normalize_backend_config(cluster_config: Dict[str, Any]) -> Dict[str, Any]: # Backend names that select a Ray backend (canonical name plus accepted aliases). _RAY_BACKEND_NAMES = frozenset({"ray", "kubernetes-ray", "ray-kubernetes", "ray_kubernetes"}) +# Stable absolute roots exported by RayBackend before a Ray Job entrypoint can +# change directory. Keep the references in one lightweight module so both the +# legacy and declarative command builders can apply identical path semantics +# without importing the Ray implementation on the default Slurm path. +RAY_JOB_CODE_DIR_ENV = "NEMO_RUN_CODE_DIR" +RAY_JOB_NEMO_SKILLS_DIR_ENV = "NEMO_SKILLS_CODE_DIR" + + +def _replace_shell_path_with_env(command: str, path: str, env_name: str) -> str: + """Replace a shell path with an expandable, quoted environment reference. + + A plain ``str.replace(path, "${VAR}")`` is incorrect when ``path`` occurs + inside single quotes because the shell would keep ``${VAR}`` literal. This + small scanner preserves the surrounding command and emits a safely quoted + reference in unquoted, double-quoted, and single-quoted contexts. + """ + result: list[str] = [] + quote: str | None = None + escaped = False + index = 0 + env_ref = f"${{{env_name}}}" + + while index < len(command): + if command.startswith(path, index): + if quote == "'": + # Close the single-quoted segment, expand in double quotes, and + # reopen it so the rest of the original segment stays literal. + result.append(f"'\"{env_ref}\"'") + elif quote == '"': + # Already protected by the surrounding double quotes. + result.append(env_ref) + else: + result.append(f'"{env_ref}"') + index += len(path) + continue + + char = command[index] + result.append(char) + + if quote == "'": + if char == "'": + quote = None + elif quote == '"': + if escaped: + escaped = False + elif char == "\\": + escaped = True + elif char == '"': + quote = None + elif escaped: + escaped = False + elif char == "\\": + escaped = True + elif char in {"'", '"'}: + quote = char + + index += 1 + + return "".join(result) + + +def rewrite_ray_job_code_paths(command: str) -> str: + """Map nemo-run package roots to stable Ray Job runtime directories. + + The more-specific NeMo-Skills path must be handled first because the Ray + working directory may contain only the calling project (for example, + NVFlow), while NeMo-Skills itself comes from the worker image's baked Python + environment. + """ + command = _replace_shell_path_with_env( + command, + "/nemo_run/code/nemo_skills", + RAY_JOB_NEMO_SKILLS_DIR_ENV, + ) + return _replace_shell_path_with_env(command, "/nemo_run/code", RAY_JOB_CODE_DIR_ENV) + def get_backend_name(cluster_config: Dict[str, Any]) -> str: """Return the normalized (lowercased) backend name, or '' when none is set. @@ -193,9 +269,12 @@ def __getattr__(name: str): "BackendRunOptions", "ExecutionBackend", "RayBackend", + "RAY_JOB_CODE_DIR_ENV", + "RAY_JOB_NEMO_SKILLS_DIR_ENV", "get_backend_name", "is_ray_backend_name", "is_ray_jobs_backend", + "rewrite_ray_job_code_paths", "get_execution_backend", "track_stage_tasks", "stop_stage_tasks", diff --git a/nemo_skills/pipeline/utils/declarative.py b/nemo_skills/pipeline/utils/declarative.py index 1dc35f2671..7974932203 100644 --- a/nemo_skills/pipeline/utils/declarative.py +++ b/nemo_skills/pipeline/utils/declarative.py @@ -31,7 +31,12 @@ run_exp, temporary_env_update, ) -from nemo_skills.pipeline.utils.backends import get_backend_name, get_execution_backend, is_ray_jobs_backend +from nemo_skills.pipeline.utils.backends import ( + get_execution_backend, + is_ray_backend_name, + is_ray_jobs_backend, + rewrite_ray_job_code_paths, +) from nemo_skills.pipeline.utils.exp import ( REUSE_CODE_EXP, get_packaging_job_key, @@ -586,27 +591,41 @@ def _rewrite_local_paths(self, script: run.Script) -> run.Script: Under the Ray Jobs API backend (``backend.name == "ray"``) the command is submitted to a *remote* cluster whose image and Python version need not match the driver's, so the driver-side absolute install path (e.g. - ``/usr/local/lib/python3.10/dist-packages``) will not exist there. In that - case rewrite to relative (cwd) paths instead -- mirroring the add_task path -- - so the command runs from the job's working directory and ``nemo_skills`` - resolves via the baked image's ``PYTHONPATH``. + ``/usr/local/lib/python3.10/dist-packages``) will not exist there. When an + explicit Ray ``working_dir`` is configured, use stable absolute roots that + the Ray entrypoint captures before the workload can change directory. Without + ``working_dir``, retain the historical cwd-relative Ray rewrite. """ nemo_repo = get_registered_external_repo("nemo_skills") # Use the shared parser so the supported bare-string form (``backend: ray``) is # handled too -- a direct ``.get("name")`` raises AttributeError on a string. - is_ray = get_backend_name(self.cluster_config) == "ray" + is_ray = is_ray_backend_name(self.cluster_config) if nemo_repo is None and not is_ray: return script + ray_jobs_with_working_dir = False if is_ray: + backend = get_execution_backend(self.cluster_config, with_ray=self.with_ray) + ray_jobs_with_working_dir = is_ray_jobs_backend(backend) and bool(getattr(backend, "working_dir", None)) + + if ray_jobs_with_working_dir: + + def _replace(cmd: str) -> str: + return rewrite_ray_job_code_paths(cmd) + + elif is_ray: pkg_path = "./nemo_skills" repo_root = "." + + def _replace(cmd: str) -> str: + return cmd.replace("/nemo_run/code/nemo_skills", pkg_path).replace("/nemo_run/code", repo_root) + else: pkg_path = str(nemo_repo.path) repo_root = str(nemo_repo.path.parent) - def _replace(cmd: str) -> str: - return cmd.replace("/nemo_run/code/nemo_skills", pkg_path).replace("/nemo_run/code", repo_root) + def _replace(cmd: str) -> str: + return cmd.replace("/nemo_run/code/nemo_skills", pkg_path).replace("/nemo_run/code", repo_root) inline_cmd = script.inline if isinstance(inline_cmd, str): diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index a3abfc025e..00bc52e777 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -33,6 +33,7 @@ get_execution_backend, is_ray_backend_name, is_ray_jobs_backend, + rewrite_ray_job_code_paths, ) from nemo_skills.pipeline.utils.cluster import ( get_env_variables, @@ -923,13 +924,22 @@ def add_server_tasks(): # no mounting here, so assuming /nemo_run/code can be replaced with the current dir if cluster_config["executor"] == "none": - # replacing /nemo_run/code/nemo_skills with the installed location - - for idx in range(len(commands)): - commands[idx] = commands[idx].replace( - "/nemo_run/code/nemo_skills", str(get_registered_external_repo("nemo_skills").path) - ) - commands[idx] = commands[idx].replace("/nemo_run/code", "./") + if ray_jobs_active and getattr(backend, "working_dir", None): + # Ray makes working_dir the initial job cwd, but workloads such as + # NeMo-RL/Gym may cd elsewhere before consuming a legacy + # /nemo_run/code path. Rewrite to absolute roots captured by the Ray + # entrypoint rather than cwd-relative paths. NeMo-Skills comes from + # the baked worker environment, not necessarily the working-dir ZIP. + for idx in range(len(commands)): + commands[idx] = rewrite_ray_job_code_paths(commands[idx]) + else: + # Preserve the historical executor:none behavior byte-for-byte when + # Ray Jobs code delivery is not explicitly configured. + for idx in range(len(commands)): + commands[idx] = commands[idx].replace( + "/nemo_run/code/nemo_skills", str(get_registered_external_repo("nemo_skills").path) + ) + commands[idx] = commands[idx].replace("/nemo_run/code", "./") first_command_image = command_images[0] if command_images else None # use_with_ray_cluster requests an EMBEDDED Ray cluster, which nemo-run only supports on diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 76922fb3fc..7a6ed24f76 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -51,7 +51,12 @@ import nemo_run as run -from nemo_skills.pipeline.utils.backends import BackendRunOptions, ExecutionBackend +from nemo_skills.pipeline.utils.backends import ( + RAY_JOB_CODE_DIR_ENV, + RAY_JOB_NEMO_SKILLS_DIR_ENV, + BackendRunOptions, + ExecutionBackend, +) from nemo_skills.utils import get_logger_name LOG = logging.getLogger(get_logger_name(__file__)) @@ -363,13 +368,41 @@ def _append_stage_job_record( exc, ) - @staticmethod - def _prepare_job_entrypoint_command(command: str) -> str: - """Rewrite a command to use local cluster autodiscovery (RAY_ADDRESS=auto).""" + def _prepare_job_entrypoint_command(self, command: str) -> str: + """Prepare local Ray discovery and stable working-directory roots.""" # When running *inside* a Ray Job, the driver is already on the cluster. # Replace any forwarded Ray Client endpoint with local cluster autodiscovery. stripped = re.sub(r"export\s+RAY_ADDRESS=[^&;]+&&\s*", "", command) - return f"export RAY_ADDRESS=auto && {stripped.strip()}" + exports = [] + if self.working_dir: + # Ray changes cwd to the uploaded working directory before invoking + # the entrypoint. Capture it before a workload can `cd` elsewhere. + exports.append(f'export {RAY_JOB_CODE_DIR_ENV}="$PWD"') + skills_ref = f"${{{RAY_JOB_NEMO_SKILLS_DIR_ENV}}}" + if skills_ref in stripped: + # The working-dir archive may intentionally contain only the + # caller (e.g. NVFlow). Resolve NeMo-Skills from the baked Python + # environment instead of assuming source was uploaded alongside it. + # Use shell globbing rather than `python -c`: a login shell may + # reset PATH even though the selected worker image contains the + # package. The ordered roots cover working-dir source and the + # baked virtualenv/system layouts used by NVFlow worker images. + exports.append( + f'export {RAY_JOB_NEMO_SKILLS_DIR_ENV}="$(for candidate in ' + f'"${{{RAY_JOB_CODE_DIR_ENV}}}/nemo_skills" ' + "/opt/NeMo-Skills/nemo_skills " + "/opt/nemo-skills/nemo_skills " + "/opt/*/.venv/lib/python*/site-packages/nemo_skills " + "/opt/*venv/lib/python*/site-packages/nemo_skills " + "/usr/local/lib/python*/dist-packages/nemo_skills " + "/usr/local/lib/python*/site-packages/nemo_skills " + "/usr/lib/python*/dist-packages/nemo_skills " + "/usr/lib/python*/site-packages/nemo_skills; " + 'do if [ -d "$candidate" ]; then printf "%s" "$candidate"; break; fi; done)"' + ) + exports.append(f'test -n "${{{RAY_JOB_NEMO_SKILLS_DIR_ENV}}}"') + exports.append("export RAY_ADDRESS=auto") + return " && ".join([*exports, stripped.strip()]) @staticmethod def _normalize_image_label_selectors( diff --git a/tests/test_backends.py b/tests/test_backends.py index 3d4e030e83..3b0d2d5812 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -12,6 +12,9 @@ # See the License for the specific language governing permissions and # limitations under the License. +import shlex +import subprocess + import pytest from nemo_skills.pipeline.utils.backends import get_backend_name, get_execution_backend, is_ray_backend_name @@ -306,6 +309,58 @@ def get_job_logs(self, job_id): } assert "pip" not in submitted[0]["runtime_env"] assert "conda" not in submitted[0]["runtime_env"] + prepared_command = shlex.split(submitted[0]["entrypoint"])[2] + assert prepared_command.startswith('export NEMO_RUN_CODE_DIR="$PWD" && export RAY_ADDRESS=auto && ') + + +def test_ray_backend_without_working_dir_preserves_entrypoint_command(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + backend = RayBackend(dashboard_url="http://ray-head:8265") + + assert backend._prepare_job_entrypoint_command("echo ok") == "export RAY_ADDRESS=auto && echo ok" + + +def test_ray_working_dir_paths_expand_after_cd_in_all_quote_contexts(tmp_path): + from nemo_skills.pipeline.utils.backends import rewrite_ray_job_code_paths + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + code_dir = tmp_path / "uploaded code" + skills_dir = code_dir / "nemo_skills" + gym_dir = tmp_path / "Gym" + code_dir.mkdir() + skills_dir.mkdir() + (skills_dir / "__init__.py").write_text("# packaged test module\n") + gym_dir.mkdir() + + command = ( + f"cd {shlex.quote(str(gym_dir))} && printf '%s\\n' " + "/nemo_run/code/nvflow/unquoted " + '"/nemo_run/code/nvflow/double quoted" ' + "'/nemo_run/code/nvflow/single quoted' " + "'/nemo_run/code/nemo_skills/dataset/test.jsonl'" + ) + rewritten = rewrite_ray_job_code_paths(command) + prepared = RayBackend( + dashboard_url="http://ray-head:8265", + working_dir="/opt/nvflow-ray-code.zip", + )._prepare_job_entrypoint_command(rewritten) + + result = subprocess.run( + ["/bin/bash", "-c", prepared], + cwd=code_dir, + check=True, + capture_output=True, + text=True, + ) + + assert result.stdout.splitlines() == [ + str(code_dir / "nvflow/unquoted"), + str(code_dir / "nvflow/double quoted"), + str(code_dir / "nvflow/single quoted"), + str(skills_dir / "dataset/test.jsonl"), + ] + assert prepared.index('export NEMO_RUN_CODE_DIR="$PWD"') < prepared.index(f"cd {shlex.quote(str(gym_dir))}") def test_ray_backend_rejects_non_string_working_dir(): diff --git a/tests/test_declarative_pipeline.py b/tests/test_declarative_pipeline.py index aaad12d9d1..28622fe6c7 100644 --- a/tests/test_declarative_pipeline.py +++ b/tests/test_declarative_pipeline.py @@ -239,6 +239,66 @@ def test_pipeline_cluster_config_passed_directly(self): class TestPipelineExecution: """Test Pipeline execution and job management.""" + @pytest.mark.parametrize("backend_name", ["ray", "kubernetes-ray", "ray-kubernetes", "ray_kubernetes"]) + def test_ray_working_dir_rewrite_survives_cd_and_quotes(self, backend_name): + cluster_config = { + "executor": "none", + "containers": {}, + "backend": { + "name": backend_name, + "dashboard_url": "http://ray-head:8265", + "working_dir": "/opt/nvflow-ray-code.zip", + }, + } + pipeline = Pipeline( + name="ray-paths", + cluster_config=cluster_config, + jobs=[{"name": "placeholder", "group": CommandGroup(commands=[make_command()], log_dir="/tmp/logs")}], + skip_hf_home_check=True, + ) + script = DummyScript( + "cd /opt/Gym && python '/nemo_run/code/nvflow/train.py' \"/nemo_run/code/nemo_skills/dataset/test.jsonl\"" + ) + + rewritten = pipeline._rewrite_local_paths(script).inline + + assert rewritten.startswith("cd /opt/Gym && ") + assert '"${NEMO_RUN_CODE_DIR}"' in rewritten + assert "${NEMO_SKILLS_CODE_DIR}" in rewritten + assert "/nemo_run/code" not in rewritten + + @patch("nemo_skills.pipeline.utils.declarative.get_registered_external_repo") + def test_declarative_non_ray_executor_none_rewrite_is_unchanged(self, mock_repo): + from pathlib import Path + from types import SimpleNamespace + + mock_repo.return_value = SimpleNamespace(path=Path("/installed/nemo_skills")) + pipeline = Pipeline( + name="local-paths", + cluster_config={"executor": "none", "containers": {}}, + jobs=[{"name": "placeholder", "group": CommandGroup(commands=[make_command()], log_dir="/tmp/logs")}], + skip_hf_home_check=True, + ) + script = DummyScript("python /nemo_run/code/nvflow/train.py /nemo_run/code/nemo_skills/dataset/test.jsonl") + + assert pipeline._rewrite_local_paths(script).inline == ( + "python /installed/nvflow/train.py /installed/nemo_skills/dataset/test.jsonl" + ) + + def test_declarative_slurm_command_is_byte_identical(self): + original = "cd /opt/Gym && python '/nemo_run/code/nvflow/train.py'" + command = make_command(inline=original) + pipeline = Pipeline( + name="slurm-paths", + cluster_config={"executor": "slurm", "containers": {}}, + jobs=[{"name": "placeholder", "group": CommandGroup(commands=[command], log_dir="/tmp/logs")}], + skip_hf_home_check=True, + ) + + script, _ = pipeline._prepare_command(command, pipeline.cluster_config) + + assert script.inline == original + @patch("nemo_skills.pipeline.utils.declarative.get_exp") @patch("nemo_skills.pipeline.utils.declarative.get_env_variables") @patch("nemo_skills.pipeline.utils.declarative.run_exp") diff --git a/tests/test_pipeline_utils.py b/tests/test_pipeline_utils.py index fdda6cf03f..6d8d7ca204 100644 --- a/tests/test_pipeline_utils.py +++ b/tests/test_pipeline_utils.py @@ -603,3 +603,82 @@ def _add(script, **kwargs): reuse_code=False, ) assert mock_resolve.call_count >= 1, "Ray Jobs add_task must resolve command images for the queue" + + +@patch("nemo_skills.pipeline.utils.exp.resolve_container_image", return_value="resolved:image") +@patch("nemo_skills.pipeline.utils.exp.get_executor") +@patch("nemo_skills.pipeline.utils.exp.get_free_port", return_value=12345) +def test_add_task_ray_working_dir_uses_stable_absolute_roots(mock_port, mock_get_executor, mock_resolve): + """A later cd must not make executor:none paths relative to /opt/Gym.""" + from types import SimpleNamespace + + from nemo_skills.pipeline.utils.exp import add_task + + mock_get_executor.return_value = MagicMock(container_image="resolved:image") + exp = SimpleNamespace(add=MagicMock(return_value="h"), jobs=[]) + original = ( + "cd /opt/Gym && python /nemo_run/code/nvflow/driver.py " + "--dataset '/nemo_run/code/nemo_skills/dataset/secque/test.jsonl'" + ) + cluster_config = { + "executor": "none", + "containers": {"sandbox": "sandbox:latest"}, + "backend": { + "name": "ray", + "dashboard_url": "http://ray-head:8265", + "working_dir": "/opt/nvflow-ray-code.zip", + }, + } + + add_task( + exp=exp, + cmd=original, + task_name="grpo-prepare", + cluster_config=cluster_config, + container="main:latest", + log_dir="/tmp/logs", + skip_hf_home_check=True, + reuse_code=False, + ) + + queued = exp._ns_ray_jobs_queue[0]["command"] + added = exp.add.call_args.args[0].inline + assert queued == added + assert queued.startswith("cd /opt/Gym && ") + assert '"${NEMO_RUN_CODE_DIR}"/nvflow/driver.py' in queued + assert '"${NEMO_SKILLS_CODE_DIR}"' in queued + assert "/nemo_run/code" not in queued + assert "./nvflow" not in queued + + +@patch("nemo_skills.pipeline.utils.exp.get_registered_external_repo") +@patch("nemo_skills.pipeline.utils.exp.get_executor") +@patch("nemo_skills.pipeline.utils.exp.get_free_port", return_value=12345) +def test_add_task_non_ray_executor_none_path_rewrite_is_unchanged(mock_port, mock_get_executor, mock_repo): + """The opt-in Ray working-dir contract must not alter legacy local execution.""" + from pathlib import Path + from types import SimpleNamespace + + from nemo_skills.pipeline.utils.exp import add_task + + mock_repo.return_value = SimpleNamespace(path=Path("/installed/nemo_skills")) + mock_get_executor.return_value = MagicMock() + exp = SimpleNamespace(add=MagicMock(return_value="h")) + + add_task( + exp=exp, + cmd=( + "cd /opt/Gym && python /nemo_run/code/nvflow/driver.py " + "--dataset /nemo_run/code/nemo_skills/dataset/test.jsonl" + ), + task_name="legacy-local", + cluster_config={"executor": "none", "containers": {"sandbox": "sandbox:latest"}}, + container="main:latest", + log_dir="/tmp/logs", + skip_hf_home_check=True, + reuse_code=False, + ) + + assert exp.add.call_args.args[0].inline == ( + "cd /opt/Gym && python .//nvflow/driver.py --dataset /installed/nemo_skills/dataset/test.jsonl" + ) From eed74e780caad36a6668ee16983ba78a81158aaa Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Fri, 11 Sep 2026 14:35:19 -0400 Subject: [PATCH 42/46] test: replace retired NVIDIA API model Signed-off-by: Nick Gupta --- docs/evaluation/external-benchmarks.md | 5 +++-- tests/test_eval.py | 5 +++-- tests/test_generation.py | 25 +++++++++++++++---------- 3 files changed, 21 insertions(+), 14 deletions(-) diff --git a/docs/evaluation/external-benchmarks.md b/docs/evaluation/external-benchmarks.md index a4c4ca4e8b..7f0acb994d 100644 --- a/docs/evaluation/external-benchmarks.md +++ b/docs/evaluation/external-benchmarks.md @@ -307,10 +307,11 @@ Run evaluation (using an API model as an example): ns eval \ --cluster=local \ --server_type=openai \ - --model=nvidia/nemotron-3-nano-30b-a3b \ + --model=nvidia/nemotron-3.5-lightning-30b-a3b \ --server_address=https://integrate.api.nvidia.com/v1 \ --benchmarks=word_count \ - --output_dir=/workspace/test-eval + --output_dir=/workspace/test-eval \ + ++inference.temperature=1.0 ``` View results: diff --git a/tests/test_eval.py b/tests/test_eval.py index 272728fca3..15474a6e9b 100644 --- a/tests/test_eval.py +++ b/tests/test_eval.py @@ -288,8 +288,8 @@ def test_eval_multi_model_generation_module_smoke(tmp_path): f"ns eval " f" --server_type=openai " f" --server_type=openai " - f" --model=nvidia/nemotron-3-nano-30b-a3b " - f" --model=nvidia/nemotron-3-nano-30b-a3b " + f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " @@ -297,6 +297,7 @@ def test_eval_multi_model_generation_module_smoke(tmp_path): f" --generation_module={shlex.quote(str(generation_module))} " f" ++max_samples=1 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) diff --git a/tests/test_generation.py b/tests/test_generation.py index 297d0b180f..be1bd60705 100644 --- a/tests/test_generation.py +++ b/tests/test_generation.py @@ -33,12 +33,13 @@ def test_eval_gsm8k_api(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-nano-30b-a3b " + f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) @@ -64,17 +65,18 @@ def test_eval_judge_api(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-nano-30b-a3b " + f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=math-500 " f" --output_dir={tmp_path} " - f" --judge_model=nvidia/nemotron-3-nano-30b-a3b " + f" --judge_model=nvidia/nemotron-3.5-lightning-30b-a3b " f" --judge_server_address=https://integrate.api.nvidia.com/v1 " f" --judge_server_type=openai " f" --judge_generation_type=math_judge " - f" --extra_judge_args='++max_concurrent_requests=1 ++inference.timeout=120 ++server.max_retries=1' " + f" --extra_judge_args='++max_concurrent_requests=1 ++inference.temperature=1.0 ++inference.timeout=120 ++server.max_retries=1' " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) @@ -100,7 +102,7 @@ def test_fail_on_api_key_env_var(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-nano-30b-a3b " + f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " @@ -122,12 +124,13 @@ def test_succeed_on_api_key_env_var(tmp_path): f"unset NVIDIA_API_KEY && " f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-nano-30b-a3b " + f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " f" ++server.api_key_env_var=MY_CUSTOM_KEY " @@ -155,12 +158,13 @@ def test_generate_openai_format(tmp_path, format): cmd = ( f"ns generate " f" --server_type=openai " - f" --model=nvidia/nemotron-3-nano-30b-a3b " + f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --input_file=/nemo_run/code/tests/data/openai-input-{format}.test " f" --output_dir={tmp_path} " f" ++prompt_format=openai " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) @@ -370,17 +374,18 @@ def test_judge_generations_with_structured_output(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-nano-30b-a3b " + f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=hle " f" --output_dir={tmp_path} " - f" --judge_model=nvidia/nemotron-3-nano-30b-a3b " + f" --judge_model=nvidia/nemotron-3.5-lightning-30b-a3b " f" --judge_server_address=https://integrate.api.nvidia.com/v1 " f" --judge_server_type=openai " f" --metric_type=hle-aa " - f' --extra_judge_args="++structured_output=HLE_JUDGE_AA ++max_concurrent_requests=1 ++inference.timeout=120 ++server.max_retries=1" ' + f' --extra_judge_args="++structured_output=HLE_JUDGE_AA ++max_concurrent_requests=1 ++inference.temperature=1.0 ++inference.timeout=120 ++server.max_retries=1" ' f" ++max_samples=2 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " f" ++inference.tokens_to_generate=1024 " # to make test go fast From 580808fa6ac17cef6d781516850b63903812deed Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Fri, 11 Sep 2026 15:02:25 -0400 Subject: [PATCH 43/46] fix(ray): reject unverifiable job dependencies Signed-off-by: Nick Gupta --- .../example-slurm-ray-k8s-precreated.yaml | 7 +- docs/basics/code-packaging.md | 6 + nemo_skills/pipeline/utils/declarative.py | 89 ++++++---- nemo_skills/pipeline/utils/exp.py | 5 +- nemo_skills/pipeline/utils/ray_backend.py | 107 ++++++++---- tests/test_backends.py | 160 +++++++++++++++--- tests/test_declarative_pipeline.py | 91 +++++++++- 7 files changed, 367 insertions(+), 98 deletions(-) diff --git a/cluster_configs/example-slurm-ray-k8s-precreated.yaml b/cluster_configs/example-slurm-ray-k8s-precreated.yaml index a305c06f77..010b1baa0e 100644 --- a/cluster_configs/example-slurm-ray-k8s-precreated.yaml +++ b/cluster_configs/example-slurm-ray-k8s-precreated.yaml @@ -16,8 +16,11 @@ # pre-created Ray cluster running on Kubernetes. # # Notes: -# - SLURM remains the scheduler used by nemo_run. -# - The execution backend is Ray; jobs connect to an existing Ray endpoint. +# - executor: slurm retains nemo-run's cluster configuration and experiment lookup. +# - The Ray Jobs backend submits directly to the existing cluster; it does not +# launch the queued commands through Slurm. +# - Same-batch dependencies are supported. For an active cross-experiment +# run_after, wait for the upstream experiment or use the default Slurm backend. # - Replace all CHANGE_ME values before use. executor: slurm diff --git a/docs/basics/code-packaging.md b/docs/basics/code-packaging.md index 63c96e52bc..e76ac8572d 100644 --- a/docs/basics/code-packaging.md +++ b/docs/basics/code-packaging.md @@ -104,3 +104,9 @@ edit YAML, JSON, or other files inside the uploaded archive. Configuration files that must refer to the delivered source after changing directory should resolve `NEMO_RUN_CODE_DIR` themselves (and may retain `/nemo_run/code` as their non-Ray default). + +The direct Ray Jobs backend can order dependencies submitted in the same batch. +It cannot observe an active job from another nemo-run or Slurm experiment, so an +active cross-experiment `run_after` fails before any Ray Job is submitted. Wait +for the upstream experiment to finish, or use the default Slurm backend when the +scheduler must enforce that dependency. diff --git a/nemo_skills/pipeline/utils/declarative.py b/nemo_skills/pipeline/utils/declarative.py index 7974932203..8931f5f4cb 100644 --- a/nemo_skills/pipeline/utils/declarative.py +++ b/nemo_skills/pipeline/utils/declarative.py @@ -442,6 +442,8 @@ def run(self, dry_run: bool = False, log_dir: Optional[str] = None, _reuse_exp=N """ # Track job name -> task handle for dependency resolution job_name_to_handle = {} + pipeline_backend = get_execution_backend(self.cluster_config) + ray_jobs_active_for_pipeline = is_ray_jobs_backend(pipeline_backend) with get_exp(self.name, self.cluster_config, _reuse_exp) as exp: # Process each job in order @@ -469,13 +471,48 @@ def run(self, dry_run: bool = False, log_dir: Optional[str] = None, _reuse_exp=N if isinstance(dep, str): # String dependency = external experiment name if self.cluster_config["executor"] == "slurm": + exp_jobs = getattr(exp, "jobs", None) + known_job_ids = ( + {getattr(job, "id", None) for job in exp_jobs} + if isinstance(exp_jobs, (list, tuple)) + else set() + ) + if _reuse_exp and dep in known_job_ids: + internal_deps.append(dep) + LOG.info(f"Job '{job_name}' depends on task handle '{dep}' (from reused experiment)") + continue + exp_handles = get_exp_handles(dep) if len(exp_handles) == 0: - LOG.warning( - f"No pending or running tasks found for experiment {dep}, cannot set dependencies." - ) - # If no experiment found, treat as direct task handle (for _reuse_exp case) - if _reuse_exp: + if ray_jobs_active_for_pipeline: + try: + all_exp_handles = get_exp_handles( + dep, + ignore_finished=False, + ignore_exp_not_exists=False, + ) + except ValueError: + all_exp_handles = [] + + if all_exp_handles: + LOG.info( + "All tasks in external experiment '%s' have already finished; " + "no dependency is needed.", + dep, + ) + else: + LOG.warning( + "No pending or running tasks found for '%s', and the dependency " + "could not be verified as a finished experiment.", + dep, + ) + # Preserve an unverifiable name for the Ray dependency + # validator. Dropping it would launch work without honoring + # the requested prerequisite. + external_deps.append(dep) + elif _reuse_exp: + # Preserve the historical reused-experiment behavior for + # the default backend, where this may be a direct task handle. internal_deps.append(dep) LOG.info( f"Job '{job_name}' depends on task handle '{dep}' (from reused experiment)" @@ -552,12 +589,14 @@ def run(self, dry_run: bool = False, log_dir: Optional[str] = None, _reuse_exp=N job_name_to_handle[job_name] = task_handle LOG.info(f"Added job '{job_name}' with task_handle={task_handle}") - # Only run if not using existing experiment (matching generate_v0.py line 331) - if not dry_run and not _reuse_exp: - run_exp(exp, self.cluster_config, sequential=sequential) + # Run or validate the complete experiment unless a caller is still + # assembling a reused experiment. run_exp keeps dry-runs offline and + # lets backends validate their queued graph without submitting work. + if not _reuse_exp: + run_exp(exp, self.cluster_config, sequential=sequential, dry_run=dry_run) # Cache experiment for code reuse in future runs - if self.cluster_config["executor"] != "none": + if not dry_run and self.cluster_config["executor"] != "none": tunnel = get_tunnel(self.cluster_config) cur_tunnel_hash = tunnel_hash(tunnel) if cur_tunnel_hash not in REUSE_CODE_EXP: @@ -965,9 +1004,10 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: ray_queue_images.append(getattr(executor, "container_image", None)) # Forward both internal (same-experiment) and external (cross-experiment - # run_after) dependencies. Dropping external deps would let Ray jobs submit - # before their prerequisites finish, since Ray ordering is resolved from - # the queued dep names rather than the nemo-run executor. + # run_after) dependencies. Already-finished external experiments were + # positively identified during dependency resolution and omitted there; + # any remaining external name is preserved so the Ray graph validator can + # fail rather than silently start too early. if ray_jobs_active: queue_ray_job_commands( exp=exp, @@ -981,31 +1021,6 @@ def _allocation_sort_key(entry: Dict) -> Tuple[int, int]: should_use_with_ray_cluster=should_use_with_ray_cluster, ) - # A run_after naming another experiment whose tasks have already finished - # resolves to empty handles and falls through as a bare experiment-name string - # in internal_deps. Only the Ray backend resolves cross-experiment ordering from - # the queued dep names (external_deps + the Ray queue), so only it needs those - # finished-cross-experiment strings dropped from exp.add(). On the default backend - # we must NOT silently drop them: nemo-run's exp.add asserts every dependency is a - # job in THIS experiment, and that fail-loud behavior is the historical contract. - # Only filter when exp.jobs is a concrete list (a real nemo-run experiment); a - # mocked or duck-typed exp leaves deps untouched so valid handles are never dropped. - exp_jobs = getattr(exp, "jobs", None) - if ray_jobs_active and internal_deps and isinstance(exp_jobs, (list, tuple)): - known_job_ids = {getattr(job, "id", None) for job in exp_jobs} - kept = [] - for dep in internal_deps: - if not isinstance(dep, str) or dep in known_job_ids: - kept.append(dep) - else: - LOG.warning( - "Dropping dependency '%s' from exp.add: not a job in this " - "experiment (cross-experiment ordering preserved via external " - "deps / Ray queue).", - dep, - ) - internal_deps = kept or None - # Add to experiment and return task ID # Note: Internal dependencies (task handles from same experiment) go to exp.add() # External dependencies (SLURM job IDs from other experiments) go to executor diff --git a/nemo_skills/pipeline/utils/exp.py b/nemo_skills/pipeline/utils/exp.py index 00bc52e777..ce1b74fd9a 100644 --- a/nemo_skills/pipeline/utils/exp.py +++ b/nemo_skills/pipeline/utils/exp.py @@ -1018,8 +1018,9 @@ def queue_ray_job_commands( """Queue commands for Ray Jobs API submission when the Ray backend is active. Both within-experiment ``task_dependencies`` and cross-experiment - ``external_dependencies`` (resolved ``run_after`` handles) are forwarded so - the Ray queue can order submissions on all declared prerequisites. + ``external_dependencies`` (resolved ``run_after`` handles) are forwarded. + The Ray backend orders observable in-batch dependencies and fails before + submission when an active cross-experiment dependency cannot be observed. """ if getattr(backend, "name", "") != "ray" or not getattr(backend, "dashboard_url", None): return 0 diff --git a/nemo_skills/pipeline/utils/ray_backend.py b/nemo_skills/pipeline/utils/ray_backend.py index 7a6ed24f76..bf59fdec94 100644 --- a/nemo_skills/pipeline/utils/ray_backend.py +++ b/nemo_skills/pipeline/utils/ray_backend.py @@ -168,38 +168,72 @@ def _compute_in_batch_dep_names(jobs: list[Dict[str, Any]]) -> set[str]: """Dependency identifiers a batch can observe to completion. The set is the nemo-run handle and the task_name of every queued job. - Jobs missing those keys contribute nothing. A dep matching this set is in - flight in the batch and must reach SUCCEEDED before its dependents start; - a dep matching none of it is gated upstream (see ``_deps_satisfied``). + Jobs missing those keys contribute nothing. Dependencies outside this set + cannot be observed by the direct Jobs API backend and must be rejected + before submission rather than silently treated as complete. """ names = {j.get("task_handle") for j in jobs if j.get("task_handle")} names |= {j.get("task_name") for j in jobs if j.get("task_name")} return names + @classmethod + def _validate_dependency_graph(cls, jobs: list[Dict[str, Any]]) -> None: + """Reject dependencies whose completion this batch cannot observe.""" + in_batch_dep_names = cls._compute_in_batch_dep_names(jobs) + unknown_dep_names = sorted( + { + dep_name + for job in jobs + for dep_name in job.get("dep_task_names") or [] + if dep_name not in in_batch_dep_names + } + ) + if unknown_dep_names: + raise NotImplementedError( + "Ray Jobs API backend cannot verify dependencies outside the current " + f"submission batch: {unknown_dep_names}. Active cross-experiment run_after " + "is not supported on this direct submission path; wait for the upstream " + "experiment to finish, or use the default Slurm backend." + ) + @staticmethod def _deps_satisfied( dep_names: list[str] | None, completed: Dict[str, str], - in_batch_dep_names: set[str], ) -> bool: """True only when every dependency is provably satisfied. - A dependency on a job submitted in this batch is satisfied only once that - job has reached SUCCEEDED (recorded under its task_name and its nemo-run - handle). It is never assumed done while still in flight, so a dependent - cannot start prematurely. A dependency that matches no job in this batch is - a prior or cross-experiment job already gated upstream and is treated as - satisfied so the resolver does not deadlock waiting on a job it can never - observe. + Dependencies are satisfied only after reaching SUCCEEDED, recorded under + either their task_name or nemo-run handle. Unknown dependencies are + rejected before the submission loop because this backend cannot observe + cross-experiment completion through the Ray Jobs API. """ - for d in dep_names or []: - if completed.get(d) == _SUCCESS_STATE: - continue - if d in in_batch_dep_names: - # In flight in this batch and not yet SUCCEEDED: keep waiting. - return False - # Not produced by this batch: gated upstream; treat as satisfied. - return True + return all(completed.get(dep_name) == _SUCCESS_STATE for dep_name in dep_names or []) + + @staticmethod + def _record_completion( + completed: Dict[str, str], + *, + task_name: str, + task_handle: str | None, + task_names_by_handle: Dict[str, set[str]], + successful_task_names_by_handle: Dict[str, set[str]], + status: str, + ) -> None: + """Record a terminal task status and complete its handle as one unit. + + A single nemo-run handle can represent multiple Ray Jobs (for example, a + server and client in one command group). The shared handle must not become + SUCCEEDED until every task represented by it has succeeded. + """ + completed[task_name] = status + if not task_handle: + return + grouped_task_names = task_names_by_handle.get(task_handle, set()) + if status == _SUCCESS_STATE: + successful_task_names_by_handle.setdefault(task_handle, set()).add(task_name) + if grouped_task_names and grouped_task_names.issubset(successful_task_names_by_handle.get(task_handle, set())): + completed[task_handle] = _SUCCESS_STATE def _get_jobs_client(self): """Return a JobSubmissionClient for the dashboard URL, raising if unavailable.""" @@ -672,6 +706,8 @@ def start_experiment(self, exp: run.Experiment, cluster_config: Dict[str, Any], if not pending_jobs: return super().start_experiment(exp, cluster_config, options) + self._validate_dependency_graph(pending_jobs) + if options.dry_run: LOG.info( "Dry run mode enabled; skipping Ray Jobs submission for %d task(s).", @@ -700,18 +736,24 @@ def _submit_jobs_concurrently(self, client, exp: run.Experiment, pending_jobs: l then submitted the judge — meaning the judge never came up while training needed it, causing the wait-for-host-file loop to time out. """ - # job_id -> status, keyed by both task_name and nemo-run handle - # so handle-named deps resolve against jobs in this batch. + # Dependency identifier -> status, keyed by task_name and nemo-run handle + # so either supported dependency form can resolve within this batch. completed: Dict[str, str] = {} # job_id -> Future futures: Dict[str, Future] = {} # remaining jobs not yet submitted pending = list(pending_jobs) - # Dependency identifiers this batch can observe to completion (handles and - # task_names of every queued job). A dep matching none of them is gated - # upstream and treated as satisfied; see _deps_satisfied / its helpers. - in_batch_dep_names = self._compute_in_batch_dep_names(pending_jobs) + # Validate defensively for callers that invoke this helper directly. The + # normal lifecycle also validates before the dry-run short circuit. + self._validate_dependency_graph(pending_jobs) + task_names_by_handle: Dict[str, set[str]] = {} + for job in pending_jobs: + task_handle = job.get("task_handle") + task_name = job.get("task_name") + if task_handle and task_name: + task_names_by_handle.setdefault(str(task_handle), set()).add(str(task_name)) + successful_task_names_by_handle: Dict[str, set[str]] = {} # job_id -> nemo-run task_handle (for recording completion under the handle name). handle_by_job_id: Dict[str, str] = {} @@ -821,7 +863,7 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: # Submit any jobs whose dependencies are now satisfied. still_pending = [] for job in pending: - if self._deps_satisfied(job.get("dep_task_names"), completed, in_batch_dep_names): + if self._deps_satisfied(job.get("dep_task_names"), completed): try: meta = _submit_one(job, idx) except Exception as exc: @@ -894,12 +936,15 @@ def _submit_one(job: Dict[str, Any], idx: int) -> Dict[str, Any]: self._handle_poll_failure(client, list(futures.keys()), exc) status = result["status"] task_name = result["task_name"] - completed[task_name] = status - # Also record completion under the nemo-run task_handle so that - # downstream jobs (whose dep_task_names are handles) can resolve. done_handle = handle_by_job_id.get(done_job_id) - if done_handle: - completed[done_handle] = status + self._record_completion( + completed, + task_name=task_name, + task_handle=done_handle, + task_names_by_handle=task_names_by_handle, + successful_task_names_by_handle=successful_task_names_by_handle, + status=status, + ) del futures[done_job_id] LOG.info("Ray job %s (%s) finished with status %s", done_job_id, task_name, status) diff --git a/tests/test_backends.py b/tests/test_backends.py index 3b0d2d5812..ffba3102eb 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -402,7 +402,7 @@ def test_default_slurm_backend_ignores_ray_working_dir_and_stays_lazy(): # --------------------------------------------------------------------------- -# Dependency resolver (in-batch vs cross-experiment deps) +# Dependency resolver # --------------------------------------------------------------------------- @@ -424,46 +424,168 @@ def test_deps_satisfied_false_for_in_batch_dep_not_yet_completed(): from nemo_skills.pipeline.utils.ray_backend import RayBackend # Premature-start guard: dep is in flight in this batch and not SUCCEEDED. - in_batch = {"train", "train-handle"} - assert RayBackend._deps_satisfied(["train"], {}, in_batch) is False + assert RayBackend._deps_satisfied(["train"], {}) is False # Recorded under a non-success terminal state still blocks. - assert RayBackend._deps_satisfied(["train"], {"train": "FAILED"}, in_batch) is False + assert RayBackend._deps_satisfied(["train"], {"train": "FAILED"}) is False def test_deps_satisfied_true_when_in_batch_dep_succeeded_by_task_name_or_handle(): from nemo_skills.pipeline.utils.ray_backend import RayBackend - in_batch = {"train", "train-handle"} # Recorded SUCCEEDED under task_name. - assert RayBackend._deps_satisfied(["train"], {"train": "SUCCEEDED"}, in_batch) is True + assert RayBackend._deps_satisfied(["train"], {"train": "SUCCEEDED"}) is True # Recorded SUCCEEDED under the nemo-run handle. - assert RayBackend._deps_satisfied(["train-handle"], {"train-handle": "SUCCEEDED"}, in_batch) is True + assert RayBackend._deps_satisfied(["train-handle"], {"train-handle": "SUCCEEDED"}) is True -def test_deps_satisfied_true_for_cross_experiment_dep_gated_upstream(): +def test_deps_satisfied_false_without_explicit_success(): from nemo_skills.pipeline.utils.ray_backend import RayBackend - # Dep matches no job in this batch -> gated upstream -> treated satisfied. - in_batch = {"judge", "judge-handle"} - assert RayBackend._deps_satisfied(["prior-experiment-handle"], {}, in_batch) is True + assert RayBackend._deps_satisfied(["prior-experiment-handle"], {}) is False -def test_deps_satisfied_mixed_in_batch_pending_and_cross_experiment(): +def test_deps_satisfied_requires_every_dependency_success(): from nemo_skills.pipeline.utils.ray_backend import RayBackend - in_batch = {"train", "train-handle"} deps = ["train", "prior-experiment-handle"] - # Blocked while the in-batch dep is still pending, even though the other is gated. - assert RayBackend._deps_satisfied(deps, {}, in_batch) is False - # Unblocks once the in-batch dep succeeds. - assert RayBackend._deps_satisfied(deps, {"train": "SUCCEEDED"}, in_batch) is True + # Every dependency needs explicit success; satisfying only one is insufficient. + assert RayBackend._deps_satisfied(deps, {}) is False + assert RayBackend._deps_satisfied(deps, {"train": "SUCCEEDED"}) is False + assert ( + RayBackend._deps_satisfied( + deps, + {"train": "SUCCEEDED", "prior-experiment-handle": "SUCCEEDED"}, + ) + is True + ) def test_deps_satisfied_true_for_empty_or_none_deps(): from nemo_skills.pipeline.utils.ray_backend import RayBackend - assert RayBackend._deps_satisfied([], {}, set()) is True - assert RayBackend._deps_satisfied(None, {}, set()) is True + assert RayBackend._deps_satisfied([], {}) is True + assert RayBackend._deps_satisfied(None, {}) is True + + +def test_shared_handle_succeeds_only_after_all_grouped_tasks_succeed(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + completed = {} + task_names_by_handle = {"nemo-run": {"server", "client"}} + successful_task_names_by_handle = {} + + RayBackend._record_completion( + completed, + task_name="server", + task_handle="nemo-run", + task_names_by_handle=task_names_by_handle, + successful_task_names_by_handle=successful_task_names_by_handle, + status="SUCCEEDED", + ) + assert completed == {"server": "SUCCEEDED"} + assert RayBackend._deps_satisfied(["nemo-run"], completed) is False + + RayBackend._record_completion( + completed, + task_name="client", + task_handle="nemo-run", + task_names_by_handle=task_names_by_handle, + successful_task_names_by_handle=successful_task_names_by_handle, + status="SUCCEEDED", + ) + assert completed["nemo-run"] == "SUCCEEDED" + assert RayBackend._deps_satisfied(["nemo-run"], completed) is True + + +def test_repeated_task_names_do_not_complete_a_later_handle_early(): + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + completed = {} + task_names_by_handle = { + "nemo-run": {"server", "client"}, + "nemo-run_1": {"server", "client"}, + } + successful_task_names_by_handle = {} + + for task_name in ("server", "client"): + RayBackend._record_completion( + completed, + task_name=task_name, + task_handle="nemo-run", + task_names_by_handle=task_names_by_handle, + successful_task_names_by_handle=successful_task_names_by_handle, + status="SUCCEEDED", + ) + assert completed["nemo-run"] == "SUCCEEDED" + + RayBackend._record_completion( + completed, + task_name="server", + task_handle="nemo-run_1", + task_names_by_handle=task_names_by_handle, + successful_task_names_by_handle=successful_task_names_by_handle, + status="SUCCEEDED", + ) + assert "nemo-run_1" not in completed + + RayBackend._record_completion( + completed, + task_name="client", + task_handle="nemo-run_1", + task_names_by_handle=task_names_by_handle, + successful_task_names_by_handle=successful_task_names_by_handle, + status="SUCCEEDED", + ) + assert completed["nemo-run_1"] == "SUCCEEDED" + + +def test_ray_backend_rejects_cross_experiment_dependency_before_submission(): + from types import SimpleNamespace + + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + class NoSubmissionClient: + def submit_job(self, **kwargs): + raise AssertionError("submit_job must not be called") + + backend = RayBackend(dashboard_url="http://ray-head:8265") + pending_jobs = [ + { + "task_name": "judge", + "task_handle": "nemo-run", + "command": "echo judge", + "dep_task_names": ["slurm://upstream/123"], + } + ] + + with pytest.raises(NotImplementedError, match="cannot verify dependencies outside the current submission batch"): + backend._submit_jobs_concurrently( + NoSubmissionClient(), + SimpleNamespace(_title="dependent-eval"), + pending_jobs, + ) + + +def test_ray_backend_dry_run_rejects_cross_experiment_dependency(): + from types import SimpleNamespace + + from nemo_skills.pipeline.utils.backends import BackendRunOptions + from nemo_skills.pipeline.utils.ray_backend import RayBackend + + backend = RayBackend(dashboard_url="http://ray-head:8265") + exp = SimpleNamespace( + _ns_ray_jobs_queue=[ + { + "task_name": "judge", + "task_handle": "nemo-run", + "command": "echo judge", + "dep_task_names": ["slurm://upstream/123"], + } + ] + ) + + with pytest.raises(NotImplementedError, match="cannot verify dependencies outside the current submission batch"): + backend.start_experiment(exp, {}, BackendRunOptions(dry_run=True)) # --------------------------------------------------------------------------- diff --git a/tests/test_declarative_pipeline.py b/tests/test_declarative_pipeline.py index 28622fe6c7..0f0f3682a5 100644 --- a/tests/test_declarative_pipeline.py +++ b/tests/test_declarative_pipeline.py @@ -785,17 +785,20 @@ def mock_get_executor(**kwargs): def test_finished_cross_experiment_dep_filtered_from_exp_add_on_ray_backend(self): """On the RAY backend, a run_after naming another experiment whose tasks already finished resolves to empty handles and falls through as a bare experiment-name - string in internal_deps (the _reuse_exp path). The Ray backend resolves - cross-experiment ordering from the queued dep names (not nemo-run handles), so that - finished-cross-exp string is dropped from exp.add while same-experiment handles are - kept. Reproduces the bug the legacy cross-experiment dependency patch fixed, now - handled natively. The default backend must NOT drop it -- see the sibling test. + string in internal_deps (the _reuse_exp path). Because no active handles remain, + the Ray backend can drop that finished-cross-exp string from exp.add while keeping + same-experiment handles. Active external handles are forwarded to the Ray queue and + rejected before submission because the direct Jobs API cannot observe them. The + default backend must NOT drop it -- see the sibling test. """ import nemo_run as run - # Empty handles -> the upstream experiment already finished / does not exist. + # The first lookup finds no active handles; the second proves the + # experiment exists and all of its tasks are finished. with patch("nemo_skills.pipeline.utils.declarative.get_exp_handles") as mock_get_handles: - mock_get_handles.return_value = [] + mock_get_handles.side_effect = lambda _dep, ignore_finished=True, **_kwargs: ( + [] if ignore_finished else ["finished-handle"] + ) with patch("nemo_skills.pipeline.utils.declarative.get_exp") as mock_get_exp: mock_exp = MagicMock(spec=run.Experiment) @@ -865,6 +868,7 @@ def mock_get_executor(**kwargs): assert mock_exp.add.call_args_list[0][1]["dependencies"] is None # job2: same-experiment handle kept, finished cross-exp string dropped. assert mock_exp.add.call_args_list[1][1]["dependencies"] == ["task_handle_1"] + assert mock_exp._ns_ray_jobs_queue[1]["dep_task_names"] == ["task_handle_1"] def test_finished_cross_experiment_dep_kept_on_default_backend(self): """The default (non-Ray) backend must NOT silently drop a finished cross-experiment @@ -940,6 +944,79 @@ def mock_get_executor(**kwargs): assert "task_handle_1" in deps assert "finished_external_experiment" in deps + def test_ray_dry_run_rejects_unverifiable_external_dependency(self): + """A user-facing dry-run must validate the Ray dependency graph.""" + import nemo_run as run + + def lookup_handles(_dep, ignore_finished=True, **_kwargs): + if ignore_finished: + return [] + raise ValueError("experiment not found") + + with patch( + "nemo_skills.pipeline.utils.declarative.get_exp_handles", + side_effect=lookup_handles, + ): + with patch("nemo_skills.pipeline.utils.declarative.get_exp") as mock_get_exp: + mock_exp = MagicMock(spec=run.Experiment) + mock_exp.__enter__ = MagicMock(return_value=mock_exp) + mock_exp.__exit__ = MagicMock(return_value=False) + mock_exp.jobs = [] + + def fake_add(*args, **kwargs): + handle = f"task_handle_{len(mock_exp.jobs) + 1}" + job = MagicMock() + job.id = handle + mock_exp.jobs.append(job) + return handle + + mock_exp.add = MagicMock(side_effect=fake_add) + mock_get_exp.return_value = mock_exp + + def mock_get_executor(**kwargs): + mock_executor = MagicMock() + mock_executor.packager = MagicMock() + mock_executor.container_image = "test/container" + return mock_executor + + with patch( + "nemo_skills.pipeline.utils.declarative.get_executor", + side_effect=mock_get_executor, + ): + command = make_command(inline="echo job", name="job") + group = CommandGroup(commands=[command], name="group", log_dir="/tmp/logs") + pipeline = Pipeline( + name="test_pipeline", + cluster_config={ + "executor": "slurm", + "backend": { + "name": "ray", + "dashboard_url": "http://ray-head:8265", + }, + "containers": {"nemo-skills": "test/container"}, + "account": "test", + "env_vars": {"HF_HOME": "/mounted/hf_home"}, + "mounts": ["/mounted/hf_home:/mounted/hf_home"], + }, + jobs=[ + { + "name": "job", + "group": group, + "dependencies": ["missing-upstream-experiment"], + } + ], + skip_hf_home_check=True, + reuse_code=False, + ) + + with pytest.raises( + NotImplementedError, + match="cannot verify dependencies outside the current submission batch", + ): + pipeline.run(dry_run=True) + + assert mock_exp._ns_ray_jobs_queue[0]["dep_task_names"] == ["missing-upstream-experiment"] + def test_run_after_dependencies_across_experiments(self, tmp_path): """Test that run_after dependencies work when chaining multiple generate/run_cmd calls. From 2b1eb92370d266f31cc727d3faffcb75c26b769b Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Fri, 11 Sep 2026 15:28:28 -0400 Subject: [PATCH 44/46] test: use entitled NVIDIA API model Signed-off-by: Nick Gupta --- docs/evaluation/external-benchmarks.md | 5 ++--- tests/test_eval.py | 5 ++--- tests/test_generation.py | 25 ++++++++++--------------- 3 files changed, 14 insertions(+), 21 deletions(-) diff --git a/docs/evaluation/external-benchmarks.md b/docs/evaluation/external-benchmarks.md index 7f0acb994d..0ecc999339 100644 --- a/docs/evaluation/external-benchmarks.md +++ b/docs/evaluation/external-benchmarks.md @@ -307,11 +307,10 @@ Run evaluation (using an API model as an example): ns eval \ --cluster=local \ --server_type=openai \ - --model=nvidia/nemotron-3.5-lightning-30b-a3b \ + --model=nvidia/nemotron-3-super-120b-a12b \ --server_address=https://integrate.api.nvidia.com/v1 \ --benchmarks=word_count \ - --output_dir=/workspace/test-eval \ - ++inference.temperature=1.0 + --output_dir=/workspace/test-eval ``` View results: diff --git a/tests/test_eval.py b/tests/test_eval.py index 15474a6e9b..b2791ea08b 100644 --- a/tests/test_eval.py +++ b/tests/test_eval.py @@ -288,8 +288,8 @@ def test_eval_multi_model_generation_module_smoke(tmp_path): f"ns eval " f" --server_type=openai " f" --server_type=openai " - f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " - f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --model=nvidia/nemotron-3-super-120b-a12b " + f" --model=nvidia/nemotron-3-super-120b-a12b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " @@ -297,7 +297,6 @@ def test_eval_multi_model_generation_module_smoke(tmp_path): f" --generation_module={shlex.quote(str(generation_module))} " f" ++max_samples=1 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) diff --git a/tests/test_generation.py b/tests/test_generation.py index be1bd60705..7f07845b10 100644 --- a/tests/test_generation.py +++ b/tests/test_generation.py @@ -33,13 +33,12 @@ def test_eval_gsm8k_api(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --model=nvidia/nemotron-3-super-120b-a12b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) @@ -65,18 +64,17 @@ def test_eval_judge_api(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --model=nvidia/nemotron-3-super-120b-a12b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=math-500 " f" --output_dir={tmp_path} " - f" --judge_model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --judge_model=nvidia/nemotron-3-super-120b-a12b " f" --judge_server_address=https://integrate.api.nvidia.com/v1 " f" --judge_server_type=openai " f" --judge_generation_type=math_judge " - f" --extra_judge_args='++max_concurrent_requests=1 ++inference.temperature=1.0 ++inference.timeout=120 ++server.max_retries=1' " + f" --extra_judge_args='++max_concurrent_requests=1 ++inference.timeout=120 ++server.max_retries=1' " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) @@ -102,7 +100,7 @@ def test_fail_on_api_key_env_var(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --model=nvidia/nemotron-3-super-120b-a12b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " @@ -124,13 +122,12 @@ def test_succeed_on_api_key_env_var(tmp_path): f"unset NVIDIA_API_KEY && " f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --model=nvidia/nemotron-3-super-120b-a12b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " f" ++server.api_key_env_var=MY_CUSTOM_KEY " @@ -158,13 +155,12 @@ def test_generate_openai_format(tmp_path, format): cmd = ( f"ns generate " f" --server_type=openai " - f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --model=nvidia/nemotron-3-super-120b-a12b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --input_file=/nemo_run/code/tests/data/openai-input-{format}.test " f" --output_dir={tmp_path} " f" ++prompt_format=openai " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) @@ -374,18 +370,17 @@ def test_judge_generations_with_structured_output(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --model=nvidia/nemotron-3-super-120b-a12b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=hle " f" --output_dir={tmp_path} " - f" --judge_model=nvidia/nemotron-3.5-lightning-30b-a3b " + f" --judge_model=nvidia/nemotron-3-super-120b-a12b " f" --judge_server_address=https://integrate.api.nvidia.com/v1 " f" --judge_server_type=openai " f" --metric_type=hle-aa " - f' --extra_judge_args="++structured_output=HLE_JUDGE_AA ++max_concurrent_requests=1 ++inference.temperature=1.0 ++inference.timeout=120 ++server.max_retries=1" ' + f' --extra_judge_args="++structured_output=HLE_JUDGE_AA ++max_concurrent_requests=1 ++inference.timeout=120 ++server.max_retries=1" ' f" ++max_samples=2 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " f" ++inference.tokens_to_generate=1024 " # to make test go fast From 3f10381da42315bf3911d5d52ee15adaf594f2df Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Fri, 11 Sep 2026 15:33:22 -0400 Subject: [PATCH 45/46] test: prefer durable NVIDIA API endpoint Signed-off-by: Nick Gupta --- docs/evaluation/external-benchmarks.md | 5 +++-- tests/test_eval.py | 5 +++-- tests/test_generation.py | 25 +++++++++++++++---------- 3 files changed, 21 insertions(+), 14 deletions(-) diff --git a/docs/evaluation/external-benchmarks.md b/docs/evaluation/external-benchmarks.md index 0ecc999339..335420870c 100644 --- a/docs/evaluation/external-benchmarks.md +++ b/docs/evaluation/external-benchmarks.md @@ -307,10 +307,11 @@ Run evaluation (using an API model as an example): ns eval \ --cluster=local \ --server_type=openai \ - --model=nvidia/nemotron-3-super-120b-a12b \ + --model=openai/gpt-oss-20b \ --server_address=https://integrate.api.nvidia.com/v1 \ --benchmarks=word_count \ - --output_dir=/workspace/test-eval + --output_dir=/workspace/test-eval \ + ++inference.temperature=1.0 ``` View results: diff --git a/tests/test_eval.py b/tests/test_eval.py index b2791ea08b..29668b03c5 100644 --- a/tests/test_eval.py +++ b/tests/test_eval.py @@ -288,8 +288,8 @@ def test_eval_multi_model_generation_module_smoke(tmp_path): f"ns eval " f" --server_type=openai " f" --server_type=openai " - f" --model=nvidia/nemotron-3-super-120b-a12b " - f" --model=nvidia/nemotron-3-super-120b-a12b " + f" --model=openai/gpt-oss-20b " + f" --model=openai/gpt-oss-20b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " @@ -297,6 +297,7 @@ def test_eval_multi_model_generation_module_smoke(tmp_path): f" --generation_module={shlex.quote(str(generation_module))} " f" ++max_samples=1 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) diff --git a/tests/test_generation.py b/tests/test_generation.py index 7f07845b10..98ec82ef6b 100644 --- a/tests/test_generation.py +++ b/tests/test_generation.py @@ -33,12 +33,13 @@ def test_eval_gsm8k_api(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-super-120b-a12b " + f" --model=openai/gpt-oss-20b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) @@ -64,17 +65,18 @@ def test_eval_judge_api(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-super-120b-a12b " + f" --model=openai/gpt-oss-20b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=math-500 " f" --output_dir={tmp_path} " - f" --judge_model=nvidia/nemotron-3-super-120b-a12b " + f" --judge_model=openai/gpt-oss-20b " f" --judge_server_address=https://integrate.api.nvidia.com/v1 " f" --judge_server_type=openai " f" --judge_generation_type=math_judge " - f" --extra_judge_args='++max_concurrent_requests=1 ++inference.timeout=120 ++server.max_retries=1' " + f" --extra_judge_args='++max_concurrent_requests=1 ++inference.temperature=1.0 ++inference.timeout=120 ++server.max_retries=1' " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) @@ -100,7 +102,7 @@ def test_fail_on_api_key_env_var(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-super-120b-a12b " + f" --model=openai/gpt-oss-20b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " @@ -122,12 +124,13 @@ def test_succeed_on_api_key_env_var(tmp_path): f"unset NVIDIA_API_KEY && " f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-super-120b-a12b " + f" --model=openai/gpt-oss-20b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " f" ++server.api_key_env_var=MY_CUSTOM_KEY " @@ -155,12 +158,13 @@ def test_generate_openai_format(tmp_path, format): cmd = ( f"ns generate " f" --server_type=openai " - f" --model=nvidia/nemotron-3-super-120b-a12b " + f" --model=openai/gpt-oss-20b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --input_file=/nemo_run/code/tests/data/openai-input-{format}.test " f" --output_dir={tmp_path} " f" ++prompt_format=openai " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " ) @@ -370,17 +374,18 @@ def test_judge_generations_with_structured_output(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=nvidia/nemotron-3-super-120b-a12b " + f" --model=openai/gpt-oss-20b " f" --server_address=https://integrate.api.nvidia.com/v1 " f" --benchmarks=hle " f" --output_dir={tmp_path} " - f" --judge_model=nvidia/nemotron-3-super-120b-a12b " + f" --judge_model=openai/gpt-oss-20b " f" --judge_server_address=https://integrate.api.nvidia.com/v1 " f" --judge_server_type=openai " f" --metric_type=hle-aa " - f' --extra_judge_args="++structured_output=HLE_JUDGE_AA ++max_concurrent_requests=1 ++inference.timeout=120 ++server.max_retries=1" ' + f' --extra_judge_args="++structured_output=HLE_JUDGE_AA ++max_concurrent_requests=1 ++inference.temperature=1.0 ++inference.timeout=120 ++server.max_retries=1" ' f" ++max_samples=2 " f" ++max_concurrent_requests=1 " + f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " f" ++inference.tokens_to_generate=1024 " # to make test go fast From 381120415cabeb9d6ba18da1b2e854ceb0465716 Mon Sep 17 00:00:00 2001 From: Nick Gupta Date: Fri, 11 Sep 2026 16:11:57 -0400 Subject: [PATCH 46/46] test: use entitled NVIDIA inference endpoint Signed-off-by: Nick Gupta --- tests/test_eval.py | 19 +++++++++----- tests/test_generation.py | 53 +++++++++++++++++++++------------------- 2 files changed, 41 insertions(+), 31 deletions(-) diff --git a/tests/test_eval.py b/tests/test_eval.py index 29668b03c5..95a1c1de7a 100644 --- a/tests/test_eval.py +++ b/tests/test_eval.py @@ -27,6 +27,10 @@ from nemo_skills.pipeline.utils import eval as eval_utils from nemo_skills.pipeline.utils.scripts import BaseJobScript, EvalClientScript +NVIDIA_TEST_API_BASE_URL = "https://inference-api.nvidia.com/v1" +NVIDIA_TEST_API_KEY_ENV_VAR = "NV_INFERENCE_API_KEY" +NVIDIA_TEST_API_MODEL = "gcp/google/gemini-2.5-flash-lite" + class FakeExp: def __enter__(self): @@ -278,7 +282,10 @@ def fake_generate(**kwargs): @pytest.mark.timeout(300) -@pytest.mark.skipif("NVIDIA_API_KEY" not in os.environ, reason="requires NVIDIA_API_KEY") +@pytest.mark.skipif( + NVIDIA_TEST_API_KEY_ENV_VAR not in os.environ, + reason=f"requires {NVIDIA_TEST_API_KEY_ENV_VAR}", +) def test_eval_multi_model_generation_module_smoke(tmp_path): repo_root = Path(__file__).resolve().parents[1] output_dir = tmp_path / "out" @@ -288,18 +295,18 @@ def test_eval_multi_model_generation_module_smoke(tmp_path): f"ns eval " f" --server_type=openai " f" --server_type=openai " - f" --model=openai/gpt-oss-20b " - f" --model=openai/gpt-oss-20b " - f" --server_address=https://integrate.api.nvidia.com/v1 " - f" --server_address=https://integrate.api.nvidia.com/v1 " + f" --model={NVIDIA_TEST_API_MODEL} " + f" --model={NVIDIA_TEST_API_MODEL} " + f" --server_address={NVIDIA_TEST_API_BASE_URL} " + f" --server_address={NVIDIA_TEST_API_BASE_URL} " f" --benchmarks=gsm8k " f" --output_dir={shlex.quote(str(output_dir))} " f" --generation_module={shlex.quote(str(generation_module))} " f" ++max_samples=1 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " + f" ++server.api_key_env_var={NVIDIA_TEST_API_KEY_ENV_VAR} " ) env = {**os.environ, "PYTHONPATH": f"{repo_root}{os.pathsep}{os.environ.get('PYTHONPATH', '')}"} subprocess.run(cmd, shell=True, check=True, env=env) diff --git a/tests/test_generation.py b/tests/test_generation.py index 98ec82ef6b..55f3036df7 100644 --- a/tests/test_generation.py +++ b/tests/test_generation.py @@ -27,21 +27,25 @@ from nemo_skills.pipeline.utils.generation import configure_client from nemo_skills.pipeline.utils.scripts import ServerScript +NVIDIA_TEST_API_BASE_URL = "https://inference-api.nvidia.com/v1" +NVIDIA_TEST_API_KEY_ENV_VAR = "NV_INFERENCE_API_KEY" +NVIDIA_TEST_API_MODEL = "gcp/google/gemini-2.5-flash-lite" + @pytest.mark.timeout(300) def test_eval_gsm8k_api(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=openai/gpt-oss-20b " - f" --server_address=https://integrate.api.nvidia.com/v1 " + f" --model={NVIDIA_TEST_API_MODEL} " + f" --server_address={NVIDIA_TEST_API_BASE_URL} " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " + f" ++server.api_key_env_var={NVIDIA_TEST_API_KEY_ENV_VAR} " ) subprocess.run(cmd, shell=True, check=True) @@ -65,20 +69,20 @@ def test_eval_judge_api(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=openai/gpt-oss-20b " - f" --server_address=https://integrate.api.nvidia.com/v1 " + f" --model={NVIDIA_TEST_API_MODEL} " + f" --server_address={NVIDIA_TEST_API_BASE_URL} " f" --benchmarks=math-500 " f" --output_dir={tmp_path} " - f" --judge_model=openai/gpt-oss-20b " - f" --judge_server_address=https://integrate.api.nvidia.com/v1 " + f" --judge_model={NVIDIA_TEST_API_MODEL} " + f" --judge_server_address={NVIDIA_TEST_API_BASE_URL} " f" --judge_server_type=openai " f" --judge_generation_type=math_judge " - f" --extra_judge_args='++max_concurrent_requests=1 ++inference.temperature=1.0 ++inference.timeout=120 ++server.max_retries=1' " + f" --extra_judge_args='++max_concurrent_requests=1 ++inference.timeout=120 ++server.max_retries=1 ++server.api_key_env_var={NVIDIA_TEST_API_KEY_ENV_VAR}' " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " + f" ++server.api_key_env_var={NVIDIA_TEST_API_KEY_ENV_VAR} " ) subprocess.run(cmd, shell=True, check=True) @@ -102,8 +106,8 @@ def test_fail_on_api_key_env_var(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=openai/gpt-oss-20b " - f" --server_address=https://integrate.api.nvidia.com/v1 " + f" --model={NVIDIA_TEST_API_MODEL} " + f" --server_address={NVIDIA_TEST_API_BASE_URL} " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " f" ++max_samples=2 " @@ -120,17 +124,16 @@ def test_fail_on_api_key_env_var(tmp_path): @pytest.mark.timeout(300) def test_succeed_on_api_key_env_var(tmp_path): cmd = ( - f"export MY_CUSTOM_KEY=$NVIDIA_API_KEY && " - f"unset NVIDIA_API_KEY && " + f"export MY_CUSTOM_KEY=${NVIDIA_TEST_API_KEY_ENV_VAR} && " + f"unset NVIDIA_API_KEY {NVIDIA_TEST_API_KEY_ENV_VAR} && " f"ns eval " f" --server_type=openai " - f" --model=openai/gpt-oss-20b " - f" --server_address=https://integrate.api.nvidia.com/v1 " + f" --model={NVIDIA_TEST_API_MODEL} " + f" --server_address={NVIDIA_TEST_API_BASE_URL} " f" --benchmarks=gsm8k " f" --output_dir={tmp_path} " f" ++max_samples=2 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " f" ++server.api_key_env_var=MY_CUSTOM_KEY " @@ -158,15 +161,15 @@ def test_generate_openai_format(tmp_path, format): cmd = ( f"ns generate " f" --server_type=openai " - f" --model=openai/gpt-oss-20b " - f" --server_address=https://integrate.api.nvidia.com/v1 " + f" --model={NVIDIA_TEST_API_MODEL} " + f" --server_address={NVIDIA_TEST_API_BASE_URL} " f" --input_file=/nemo_run/code/tests/data/openai-input-{format}.test " f" --output_dir={tmp_path} " f" ++prompt_format=openai " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " + f" ++server.api_key_env_var={NVIDIA_TEST_API_KEY_ENV_VAR} " ) subprocess.run(cmd, shell=True, check=True) @@ -374,20 +377,20 @@ def test_judge_generations_with_structured_output(tmp_path): cmd = ( f"ns eval " f" --server_type=openai " - f" --model=openai/gpt-oss-20b " - f" --server_address=https://integrate.api.nvidia.com/v1 " + f" --model={NVIDIA_TEST_API_MODEL} " + f" --server_address={NVIDIA_TEST_API_BASE_URL} " f" --benchmarks=hle " f" --output_dir={tmp_path} " - f" --judge_model=openai/gpt-oss-20b " - f" --judge_server_address=https://integrate.api.nvidia.com/v1 " + f" --judge_model={NVIDIA_TEST_API_MODEL} " + f" --judge_server_address={NVIDIA_TEST_API_BASE_URL} " f" --judge_server_type=openai " f" --metric_type=hle-aa " - f' --extra_judge_args="++structured_output=HLE_JUDGE_AA ++max_concurrent_requests=1 ++inference.temperature=1.0 ++inference.timeout=120 ++server.max_retries=1" ' + f' --extra_judge_args="++structured_output=HLE_JUDGE_AA ++max_concurrent_requests=1 ++inference.timeout=120 ++server.max_retries=1 ++server.api_key_env_var={NVIDIA_TEST_API_KEY_ENV_VAR}" ' f" ++max_samples=2 " f" ++max_concurrent_requests=1 " - f" ++inference.temperature=1.0 " f" ++inference.timeout=120 " f" ++server.max_retries=1 " + f" ++server.api_key_env_var={NVIDIA_TEST_API_KEY_ENV_VAR} " f" ++inference.tokens_to_generate=1024 " # to make test go fast ) subprocess.run(cmd, shell=True, check=True)