Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
46 commits
Select commit Hold shift + click to select a range
ceea2c3
feat(backend): add a Ray Jobs API execution backend
nicgupta-nvidia Jun 1, 2026
569cc15
feat(backend): support the Ray Jobs backend in generate.py and cli.py
nicgupta-nvidia Jun 2, 2026
0149074
fix(ray-backend): explicit entrypoint_label_selector keys take preced…
nicgupta-nvidia Jun 3, 2026
23e1038
fix(ray-backend): resolve cross-experiment dependency deadlock
nicgupta-nvidia Jun 9, 2026
9c53e16
feat(ray-backend): forward cluster env vars to each Ray job's runtime…
nicgupta-nvidia Jun 9, 2026
3748a8a
style: apply ruff format
nicgupta-nvidia Jun 9, 2026
b6cb822
docs: note NVFlow 2-cluster routing in the precreated-Ray sample config
nicgupta-nvidia Jun 9, 2026
c758108
fix(ray-backend): forward external deps and require proven completion
nicgupta-nvidia Jun 9, 2026
9408b56
fix(backend): treat dashboard_url as precreated signal; narrow exp lo…
nicgupta-nvidia Jun 9, 2026
c0aa065
docs(config): move sample preflight block to top-level config scope
nicgupta-nvidia Jun 9, 2026
2c596b5
test(ray-backend): cover dependency resolver and poll-failure cleanup
nicgupta-nvidia Jun 9, 2026
dc67764
fix(backend): honor dashboard_url precreated clusters in metadata and…
nicgupta-nvidia Jun 9, 2026
d3b2593
fix(ray-backend): honor use_with_ray_cluster for non-precreated stage…
nicgupta-nvidia Jun 9, 2026
9026523
docs(backend): add docstrings to Ray backend and execution-backend APIs
nicgupta-nvidia Jun 9, 2026
169ce82
fix(backend): accept dashboard_url for precreated Ray preflight
nicgupta-nvidia Jun 10, 2026
a7a61ef
fix(declarative): filter finished cross-experiment deps out of exp.add
nicgupta-nvidia Jun 10, 2026
f8697ec
chore(backends): remove dead _RayBackendShim re-export shim
nicgupta-nvidia Jun 10, 2026
0057fc2
chore(ci): re-trigger CI
nicgupta-nvidia Jun 10, 2026
e40ace1
fix(ray): merge job + driver runtime_env via RAY_OVERRIDE_JOB_RUNTIME…
nicgupta-nvidia Jun 11, 2026
98f8f47
test(ray): update _build_runtime_env tests for RAY_OVERRIDE_JOB_RUNTI…
nicgupta-nvidia Jun 12, 2026
be5a64f
feat(ray): fail fast when backend.name=ray + executor=none lacks a da…
nicgupta-nvidia Jun 12, 2026
a351194
fix(ray): use relative code paths in declarative rewrite under Ray ba…
nicgupta-nvidia Jun 15, 2026
80d9c91
fix(ray-backend): only pass entrypoint_label_selector when set and su…
nicgupta-nvidia Jun 24, 2026
fb18e02
fix(ray): keep the Ray Jobs backend off the non-Ray code path (review)
nicgupta-nvidia Jun 26, 2026
a0760f8
test(ray): assert use_with_ray_cluster key absence, not falsiness (re…
nicgupta-nvidia Jun 30, 2026
30782d9
fix(ray): avoid duplicate logs in reused experiments
nicgupta-nvidia Aug 4, 2026
bb09982
fix: isolate NLTK data download during image build
nicgupta-nvidia Aug 5, 2026
1fea337
fix: pin NLTK for the Python 3.10 image
nicgupta-nvidia Aug 5, 2026
0f1812e
fix(ray): preserve Slurm execution semantics
nicgupta-nvidia Aug 5, 2026
f25abc3
fix(security): floor litellm/wandb/lxml past known CVEs; relax stale …
nicgupta-nvidia Jul 9, 2026
e16936e
📝 CodeRabbit Chat: Add unit tests for PR changes
coderabbitai[bot] Jul 9, 2026
a800cd0
📝 CodeRabbit Chat: Add unit tests for PR changes
coderabbitai[bot] Jul 9, 2026
8eb4571
test: consolidate generated dependency-pin tests into one suite
nicgupta-nvidia Jul 9, 2026
92d1582
test(security): functional tests that NeMo-Skills works with the bump…
nicgupta-nvidia Jul 13, 2026
2e0184a
fix(security): remediate nemo-skills high findings
nicgupta-nvidia Jul 30, 2026
b866cf0
chore: satisfy requirements sorting hook
nicgupta-nvidia Jul 30, 2026
bfee735
fix(container): validate patched wandb core help output
nicgupta-nvidia Jul 30, 2026
faebf36
fix(container): remove uv Git cache from runtime image
nicgupta-nvidia Jul 30, 2026
bcfcfaf
fix(security): floor msgpack and setuptools
nicgupta-nvidia Jul 31, 2026
dcf41f6
fix(ray): deliver baked source with working directory
nicgupta-nvidia Aug 5, 2026
982bc66
fix(ray): preserve delivered code roots after cd
nicgupta-nvidia Aug 5, 2026
eed74e7
test: replace retired NVIDIA API model
nicgupta-nvidia Sep 11, 2026
580808f
fix(ray): reject unverifiable job dependencies
nicgupta-nvidia Sep 11, 2026
2b1eb92
test: use entitled NVIDIA API model
nicgupta-nvidia Sep 11, 2026
3f10381
test: prefer durable NVIDIA API endpoint
nicgupta-nvidia Sep 11, 2026
3811204
test: use entitled NVIDIA inference endpoint
nicgupta-nvidia Sep 11, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
128 changes: 128 additions & 0 deletions cluster_configs/example-slurm-ray-k8s-precreated.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,128 @@
# Copyright (c) 2026, NVIDIA CORPORATION. All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

# This config is a sample for running NeMo Skills and NVFlow recipes against a
# pre-created Ray cluster running on Kubernetes.
#
# Notes:
# - executor: slurm retains nemo-run's cluster configuration and experiment lookup.
# - The Ray Jobs backend submits directly to the existing cluster; it does not
# launch the queued commands through Slurm.
# - Same-batch dependencies are supported. For an active cross-experiment
# run_after, wait for the upstream experiment or use the default Slurm backend.
# - Replace all CHANGE_ME values before use.

executor: slurm

containers:
# Point to images available on your SLURM cluster.
trtllm: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc8
vllm: nvcr.io/nvidia/pytorch:25.02-py3
sglang: lmsysorg/sglang:v0.5.10.post1
megatron: nvcr.io/nvidia/pytorch:25.02-py3
sandbox: nvcr.io/nvidia/pytorch:25.02-py3
nemo-skills: nvcr.io/nvidia/pytorch:25.02-py3
verl: nvcr.io/nvidia/pytorch:25.02-py3
nemo-rl: nvcr.io/nvidia/pytorch:25.02-py3

job_name_prefix: "nemo_skills:"

# If launching from outside cluster, configure ssh_tunnel and remove top-level job_dir.
# ssh_tunnel:
# host: CHANGE_ME.cluster.company.com
# user: CHANGE_ME
# job_dir: /lustre/CHANGE_ME/nemo_run/jobs
# identity: /home/CHANGE_ME/.ssh/id_rsa

# If launching from within cluster, use job_dir directly.
job_dir: /lustre/CHANGE_ME/nemo_run/jobs

account: CHANGE_ME_ACCOUNT
partition: CHANGE_ME_GPU_PARTITION
cpu_partition: CHANGE_ME_CPU_PARTITION

# Optional timeout mapping by partition.
default_timeout: "2-00:00:00"
timeouts:
CHANGE_ME_GPU_PARTITION: "2-00:00:00"
CHANGE_ME_CPU_PARTITION: "1-00:00:00"

# Backend config: pre-created Ray cluster on Kubernetes.
# This sample targets ONE cluster. For NVFlow's 2-cluster (CPU + GPU) setup that
# auto-routes each stage by its num_gpus, see NVFlow's
# docs/recipes/finance/install-ray.md (backend.dashboard_url + cpu_dashboard_url).
backend:
name: ray
control_plane: kubernetes
precreated_cluster: true
# Ray Client endpoint to connect to existing cluster.
endpoint: ray://ray-head.ray.svc.cluster.local:10001
kubernetes:
# offline: batch style, online: service style lifecycle hints
mode: offline
# Optional labels forwarded to Ray job entrypoint placement for pre-created clusters.
# Example equivalent in Ray client API:
# client.submit_job(..., entrypoint_label_selector={"type": "worker"})
# entrypoint_label_selector:
# type: worker
# Optionally append a label derived from the selected NeMo Skills container image.
# image_label_key: nemo/image
# Optionally provide explicit labels keyed by container name from cluster_config.containers.
# These labels are merged into entrypoint_label_selector for ray job submit.
# image_label_selectors:
# nemo-skills:
# nemo/workload: skills
# nemo-rl:
# key: nemo/workload
# value: rl

# Optional preflight checks run before Ray-backed experiment start.
# Read from the TOP LEVEL of the cluster config (a sibling of `backend:`), not
# from under `backend:`. Useful for failing fast when endpoint, labels, or
# cluster capacity are not aligned with the job requirements.
# preflight:
# enabled: true
# # If true, require Ray endpoint connectivity and at least one live node.
# require_ray_endpoint: true
# # If true, fail if label inspection cannot be performed.
# strict_label_check: true
# # Require at least one live node matching each key/value pair.
# required_node_labels:
# - key: nemo/has-nemo-rl
# value: "true"
# - key: workload
# value: "training"
# # Require minimum total live-cluster resources before submission.
# # Supported aliases: gpu|gpus -> GPU, cpu|cpus -> CPU, mem|memory -> memory.
# min_cluster_resources:
# gpu: 16
# cpu: 64

# Required mounts for models/data/workspace.
mounts:
- /lustre/CHANGE_ME/models:/models
- /lustre/CHANGE_ME/data:/data
- /lustre/CHANGE_ME/workspace:/workspace

# HF_HOME must be mounted when skip_hf_home_check is false (default).
env_vars:
- HF_HOME=/models/hf-cache
# Optional but often useful for large checkpoint IO.
- NCCL_DEBUG=warn
- TORCH_DISTRIBUTED_DEBUG=off

# Optional extra vars to pass through from launcher environment:
# required_env_vars:
# - WANDB_API_KEY
# - NVIDIA_API_KEY
2 changes: 1 addition & 1 deletion core/pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@

[build-system]
requires = [
"setuptools",
"setuptools>=78.1.1",
"wheel"
]
build-backend = "setuptools.build_meta"
Expand Down
10 changes: 7 additions & 3 deletions core/requirements.txt
Original file line number Diff line number Diff line change
Expand Up @@ -6,13 +6,17 @@
bs4
compute-eval @ git+https://github.com/NVIDIA/compute-eval.git@e01a5d2
contractions
# Fixes eight High findings through CVE-2026-55415.
datamodel-code-generator>=0.64.0
datasets
editdistance
evalplus @ git+https://github.com/evalplus/evalplus@c91370f
faiss-cpu
fire
flask
func-timeout
# Fixes six High GitPython advisories through GHSA-94p4-4cq8-9g67.
GitPython>=3.1.55
gradio
httpx
huggingface_hub
Expand Down Expand Up @@ -47,6 +51,6 @@ sympy
torchcodec
tqdm
transformers
# Floors click to >=8.2 (wandb 0.26.x only requires click>=8.0.1, so the resolver
# was settling on click 8.1.8). Taken from #1507.
wandb>=0.27.1
# 0.28.1 is paired with the patched wandb-core built in Dockerfile.nemo-skills.
# Pinning keeps the Python/core protocol pair deterministic.
wandb==0.28.1
51 changes: 47 additions & 4 deletions dockerfiles/Dockerfile.nemo-skills
Original file line number Diff line number Diff line change
@@ -1,5 +1,31 @@
# W&B 0.28.1 still bundles Go 1.26.4, grpc-go 1.82.0, and x/text 0.38.0.
# W&B commit e118409 is its focused July 17 dependency update to Go 1.26.5,
# grpc-go 1.82.1, and x/text 0.40.0. Build only wandb-core here so the final
# image gets the fixed binary without retaining the Go toolchain or source.
ARG WANDB_CORE_COMMIT=e1184091520c9b44aa1096fdb27b2f4bf52f26d7
FROM golang:1.26.5 AS wandb-core-builder
ARG WANDB_CORE_COMMIT
RUN git init /src/wandb && \
cd /src/wandb && \
git remote add origin https://github.com/wandb/wandb.git && \
git sparse-checkout init --cone && \
git sparse-checkout set core && \
git fetch --depth 1 origin "${WANDB_CORE_COMMIT}" && \
git checkout --detach FETCH_HEAD
RUN cd /src/wandb/core && \
CGO_ENABLED=0 go build \
-tags "disable_grpc_modules parquet_read_only" \
-ldflags "-s -w -X main.commit=${WANDB_CORE_COMMIT}" \
-mod=vendor \
-o /wandb-core \
./cmd/wandb-core && \
go version -m /wandb-core | grep -F "go1.26.5" && \
go version -m /wandb-core | grep -E "google\.golang\.org/grpc[[:space:]]+v1\.82\.1([[:space:]]|$)" && \
go version -m /wandb-core | grep -E "golang\.org/x/text[[:space:]]+v0\.40\.0([[:space:]]|$)"

# using ubuntu instead of debian for easier apptainer installation on arm64
FROM ubuntu:22.04
ARG WANDB_CORE_COMMIT

# Install Python and other dependencies
RUN apt-get update && \
Expand All @@ -14,7 +40,7 @@ RUN apt-get update && \
ln -s /usr/bin/python3 /usr/bin/python && \
rm -rf /var/cache/apt/archives /var/lib/apt/lists/*

RUN pip install --upgrade pip setuptools "uv>=0.11.10"
RUN pip install --upgrade pip "setuptools>=78.1.1" "uv>=0.11.10"

# Update package lists and install apptainer for arm64
# https://apptainer.org/docs/admin/1.1/installation.html
Expand Down Expand Up @@ -59,8 +85,9 @@ RUN cd ${IFBENCH_DIR} && pip install -r requirements.txt
COPY dockerfiles/ifbench.patch /opt/benchmarks/IFBench/ifbench.patch
RUN cd /opt/benchmarks/IFBench && git apply ifbench.patch

RUN pip install langdetect absl-py immutabledict nltk ipython && \
python -c "import nltk; from spacy.cli import download; nltk.download('punkt'); nltk.download('punkt_tab'); \
# NLTK 3.10's import-safety hook is incompatible with this image's Python 3.10 runtime.
RUN pip install langdetect absl-py immutabledict "nltk==3.9.2" ipython && \
python -I -c "import nltk; from spacy.cli import download; nltk.download('punkt'); nltk.download('punkt_tab'); \
nltk.download('stopwords'); nltk.download('averaged_perceptron_tagger_eng'); download('en_core_web_sm')"

# we aren't copying main nemo_skills folder as it will always be mounted from host
Expand All @@ -73,9 +100,25 @@ COPY core/requirements.txt /opt/NeMo-Skills/core/requirements.txt
RUN pip install git+https://github.com/NVIDIA/NeMo-speech-data-processor@29b9b1ec0ceaf3ffa441c1d01297371b3f8e11d2
ARG CACHEBUST=4
# Install via `uv pip` from the project directory so [tool.uv].override-dependencies
# in pyproject.toml (which relaxes leptonai's httpx==0.27.2 pin so litellm 1.83.x
# in pyproject.toml (which relaxes leptonai's httpx==0.27.2 pin so litellm 1.84.x
# can be installed) is picked up. Plain pip ignores [tool.uv] and the resolver fails.
RUN cd /opt/NeMo-Skills && uv pip install --system --no-cache-dir \
-r core/requirements.txt -r requirements/pipeline.txt
# Replace W&B's vulnerable release binary with the source-compatible patched core
# built and module-verified above. The copy preserves its executable mode.
COPY --from=wandb-core-builder /wandb-core /usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core
RUN /usr/local/lib/python3.10/dist-packages/wandb/bin/wandb-core --help 2>&1 | \
grep -F "Commit SHA: ${WANDB_CORE_COMMIT}"
# Fix http mismatch between lepton and dggs by manually downloading dggs here
RUN pip install ddgs

# Guard the final resolved environment against the two High findings seen in
# the July 31 multi-architecture image scan.
RUN python -c "from importlib.metadata import version as v; from packaging.version import Version as V; \
assert V(v('msgpack')) >= V('1.2.1'), v('msgpack'); \
assert V(v('setuptools')) >= V('78.1.1'), v('setuptools'); \
print('msgpack/setuptools security floors OK')"

# nSpect's global policy flags Git metadata left by uv's source-distribution
# cache. The cache is build-only, so remove it from the published image.
RUN rm -rf /root/.cache/uv
42 changes: 42 additions & 0 deletions docs/basics/code-packaging.md
Original file line number Diff line number Diff line change
Expand Up @@ -68,3 +68,45 @@ If you want to have more fine-grained control over code reuse, you can directly
While our job submission is somewhat complicated and goes through NeMo-Run, at the end, we simply execute a particular sbatch file
that is uploaded to the cluster. It is helpful sometimes to see what's in it and modify directly. You can find sbatch file(s)
for each job inside `ssh_tunnel.job_dir` cluster folder that is defined in your cluster config.

## Ray Jobs code delivery

The Ray Jobs backend can optionally deliver source code with Ray's native
`runtime_env.working_dir` packaging:

```yaml
executor: none
backend:
name: ray
dashboard_url: http://<ray-head>:8265
working_dir: /opt/my-project # local directory or local .zip on the submitter
```

When `working_dir` is set, Ray uploads that directory or archive and makes it
the submitted job's current directory. It does not run `pip`, `conda`, or `uv`,
so every dependency must already be present in the Ray worker image. For a
strict-airgap launch, point it at an immutable source directory or archive baked
into the launcher image. If the option is absent, Ray Jobs retain their existing
behavior and no working directory is delivered.

Before starting the command, the backend captures Ray's initial uploaded-code
directory in `NEMO_RUN_CODE_DIR`. Legacy `/nemo_run/code/...` command paths are
rewritten to that absolute root, so they continue to work if a workload later
changes directory (for example, `cd /opt/Gym`). Legacy
`/nemo_run/code/nemo_skills/...` paths resolve separately from the NeMo-Skills
package baked into the worker image; the working-directory archive does not need
to duplicate that package. These rewrites apply only to Ray Jobs with an explicit
`working_dir`; local `executor: none`, embedded Ray-on-Slurm, and ordinary Slurm
execution keep their existing path and packaging behavior.

The backend rewrites paths in generated entrypoint command strings. It does not
edit YAML, JSON, or other files inside the uploaded archive. Configuration files
that must refer to the delivered source after changing directory should resolve
`NEMO_RUN_CODE_DIR` themselves (and may retain `/nemo_run/code` as their non-Ray
default).

The direct Ray Jobs backend can order dependencies submitted in the same batch.
It cannot observe an active job from another nemo-run or Slurm experiment, so an
active cross-experiment `run_after` fails before any Ray Job is submitted. Wait
for the upstream experiment to finish, or use the default Slurm backend when the
scheduler must enforce that dependency.
5 changes: 3 additions & 2 deletions docs/evaluation/external-benchmarks.md
Original file line number Diff line number Diff line change
Expand Up @@ -307,10 +307,11 @@ Run evaluation (using an API model as an example):
ns eval \
--cluster=local \
--server_type=openai \
--model=nvidia/nemotron-3-nano-30b-a3b \
--model=openai/gpt-oss-20b \
--server_address=https://integrate.api.nvidia.com/v1 \
--benchmarks=word_count \
--output_dir=/workspace/test-eval
--output_dir=/workspace/test-eval \
++inference.temperature=1.0
```

View results:
Expand Down
3 changes: 2 additions & 1 deletion nemo_skills/inference/eval/bfcl.py
Original file line number Diff line number Diff line change
Expand Up @@ -66,7 +66,8 @@
"cohere==5.18.0",
"typer>=0.12.5",
"tabulate>=0.9.0",
"datamodel-code-generator==0.25.7",
# 0.64.0 fixes eight High findings through CVE-2026-55415.
"datamodel-code-generator==0.64.0",
"google-genai>=1.52.0",
# "qwen-agent", # disabling due to some issues (and shouldn't be needed)
"mpmath==1.3.0",
Expand Down
6 changes: 5 additions & 1 deletion nemo_skills/pipeline/generate.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@
from nemo_skills.dataset.utils import import_from_path
from nemo_skills.inference import GENERATION_MODULE_MAP, GenerationType
from nemo_skills.pipeline.app import app, typer_unpacker
from nemo_skills.pipeline.utils.backends import get_backend_name
from nemo_skills.pipeline.utils.cluster import parse_kwargs
from nemo_skills.pipeline.utils.declarative import (
Command,
Expand Down Expand Up @@ -649,6 +650,8 @@ def convert_server_type_to_string(server_type):
if not jobs:
return None

with_ray_pipeline = get_backend_name(cluster_config) == "ray"

# Create and run pipeline
pipeline = Pipeline(
name=expname,
Expand All @@ -657,10 +660,11 @@ def convert_server_type_to_string(server_type):
reuse_code=reuse_code,
reuse_code_exp=reuse_code_exp,
skip_hf_home_check=skip_hf_home_check,
with_ray=with_ray_pipeline,
)

# TODO: remove after https://github.com/NVIDIA-NeMo/Skills/issues/578 is resolved as default will be single job
sequential = True if cluster_config["executor"] in ["local", "none"] else False
sequential = True if cluster_config["executor"] in ["local", "none"] and not with_ray_pipeline else False

# Pass _reuse_exp to pipeline.run() to add jobs to existing experiment
result = pipeline.run(dry_run=dry_run, _reuse_exp=_reuse_exp, sequential=sequential)
Expand Down
7 changes: 7 additions & 0 deletions nemo_skills/pipeline/utils/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,13 @@

# importing every utility function here to make them available in the pipeline.utils namespace

from nemo_skills.pipeline.utils.backends import (
BackendRunOptions,
ExecutionBackend,
get_execution_backend,
stop_stage_tasks,
track_stage_tasks,
)
from nemo_skills.pipeline.utils.cluster import (
_get_tunnel_cached,
cluster_download_dir,
Expand Down
Loading
Loading