Skip to content
Open
Show file tree
Hide file tree
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion deepspeed/env_report.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
import subprocess
import argparse
from .ops.op_builder.all_ops import ALL_OPS
from .ops.op_builder.builder import probe_is_compatible
from .git_version_info import installed_ops, torch_info, accelerator_name
from deepspeed.accelerator import get_accelerator

Expand Down Expand Up @@ -51,7 +52,7 @@ def op_report(verbose=True):
no = f"{YELLOW}[NO]{END}"
for op_name, builder in ALL_OPS.items():
dots = "." * (max_dots - len(op_name))
is_compatible = OKAY if builder.is_compatible(verbose) else no
is_compatible = OKAY if probe_is_compatible(builder, verbose) else no
is_installed = installed if installed_ops.get(op_name,
False) and accelerator_name == get_accelerator()._name else no
dots2 = '.' * ((len(h[1]) + (max_dots2 - len(h[1]))) - (len(is_installed) - color_len))
Expand Down
3 changes: 2 additions & 1 deletion deepspeed/git_version_info.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,9 +23,10 @@

# compatible_ops list is recreated for each launch
from .ops.op_builder.all_ops import ALL_OPS
from .ops.op_builder.builder import probe_is_compatible

compatible_ops = dict.fromkeys(ALL_OPS.keys(), False)
for op_name, builder in ALL_OPS.items():
op_compatible = builder.is_compatible()
op_compatible = probe_is_compatible(builder)
compatible_ops[op_name] = op_compatible
compatible_ops["deepspeed_not_implemented"] = False
23 changes: 22 additions & 1 deletion op_builder/builder.py
Original file line number Diff line number Diff line change
Expand Up @@ -50,7 +50,13 @@ def installed_cuda_version(name=""):
if cuda_home is None:
raise MissingCUDAException("CUDA_HOME does not exist, unable to compile CUDA op(s)")
# Ensure there is not a cuda version mismatch between torch and nvcc compiler
output = subprocess.check_output([cuda_home + "/bin/nvcc", "-V"], universal_newlines=True)
nvcc = cuda_home + "/bin/nvcc"
Comment thread
delock marked this conversation as resolved.
try:
output = subprocess.check_output([nvcc, "-V"], universal_newlines=True)
except (OSError, subprocess.SubprocessError) as err:
# CUDA_HOME can point at a runtime only install, which has no nvcc to compile with. Report that
# the same way as a missing CUDA_HOME so the callers that already fall back can handle it.
raise MissingCUDAException(f"Unable to run {nvcc}, unable to compile CUDA op(s): {err}") from err
output_split = output.split()
release_idx = output_split.index("release")
release = output_split[release_idx + 1].replace(',', '').split(".")
Expand All @@ -59,6 +65,21 @@ def installed_cuda_version(name=""):
return int(cuda_major), int(cuda_minor)


def probe_is_compatible(builder, verbose=False):
"""Report whether an op can be built, treating a probe that itself fails as "not compatible".

Compatibility is probed for every op when deepspeed is imported, whether or not the caller will ever
build that op. A CUDA op builder on a machine that has a GPU but no CUDA toolkit raises
MissingCUDAException from its probe, which would otherwise make importing deepspeed fail outright
over an op that was never asked for. See #7452.
"""
try:
return builder.is_compatible(verbose)
except Exception as err:
print(f"{WARNING} {builder.name} compatibility check failed ({err}), treating the op as not compatible")
return False


def get_default_compute_capabilities():
compute_caps = DEFAULT_COMPUTE_CAPABILITIES
# Update compute capability according to: https://en.wikipedia.org/wiki/CUDA#GPUs_supported
Expand Down
40 changes: 40 additions & 0 deletions tests/unit/ops/test_op_builder.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,9 @@

BUILDER_MODULE = builder_module
CUDA_API = BUILDER_MODULE.torch.cuda #ignore-cuda
MissingCUDAException = builder_module.MissingCUDAException
installed_cuda_version = builder_module.installed_cuda_version
probe_is_compatible = builder_module.probe_is_compatible


class _StubCUDAOpBuilder(CUDAOpBuilder):
Expand Down Expand Up @@ -298,3 +301,40 @@ def test_forked_child_can_use_cuda_after_importing_deepspeed():
pytest.skip("no CUDA device available")
assert result.returncode == 0, ("forked child could not use CUDA after 'import deepspeed' "
"(a CUDA context was created during import, issue #7918):\n" + result.stderr)


def test_installed_cuda_version_reports_a_runtime_only_cuda_home_as_missing(tmp_path):
# CUDA_HOME often points at a runtime only install (a pip torch wheel, a conda cudatoolkit) that has
# no nvcc under bin/. That has to surface as MissingCUDAException, because that is what the callers
# falling back to a CPU only build already catch. See #7452.
with patch("torch.utils.cpp_extension.CUDA_HOME", str(tmp_path)):
with pytest.raises(MissingCUDAException, match="unable to compile CUDA op"):
installed_cuda_version()


class _RaisingBuilder:
name = "raising_stub"

def is_compatible(self, verbose=False):
raise MissingCUDAException("CUDA_HOME does not exist, unable to compile CUDA op(s)")


class _AnsweringBuilder:

def __init__(self, answer):
self.name = "answering_stub"
self.answer = answer

def is_compatible(self, verbose=False):
return self.answer


def test_probe_is_compatible_does_not_propagate_a_failed_probe():
# Every op is probed at "import deepspeed" time whether or not it will ever be built, so one op
# builder that cannot answer must not take the whole import down.
assert probe_is_compatible(_RaisingBuilder()) is False


@pytest.mark.parametrize("answer", [True, False])
def test_probe_is_compatible_passes_through_a_successful_probe(answer):
assert probe_is_compatible(_AnsweringBuilder(answer)) is answer
Loading