Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions src/nemo/lens/resources/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@
publish_otel_resource_attributes,
set_otel_resource_attributes,
)
from nemo.lens.resources.gpu import detect_gpu
from nemo.lens.resources.kubernetes import detect_kubernetes
from nemo.lens.resources.local import detect_local
from nemo.lens.resources.slurm import detect_slurm
Expand All @@ -46,4 +47,5 @@ def detect_resource() -> dict:
"extend_otel_resource_attributes",
"set_otel_resource_attributes",
"publish_otel_resource_attributes",
"detect_gpu",
]
155 changes: 155 additions & 0 deletions src/nemo/lens/resources/gpu.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,155 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

"""GPU identity detection."""

from __future__ import annotations

import os
from collections.abc import Mapping
from typing import TYPE_CHECKING

from nemo.lens.semconv import (
NV_GPU_COMPUTE_CAPABILITY,
NV_GPU_DRIVER_VERSION,
NV_GPU_INDEX,
NV_GPU_MEMORY_TOTAL,
NV_GPU_MODEL,
NV_GPU_PCI_BUS_ID,
NV_GPU_SERIAL,
NV_GPU_UUID,
)

if TYPE_CHECKING:
from nemo.lens.resources.attributes import ResourceAttributeValue


def detect_gpu(
local_rank: int | str | None = None,
environ: Mapping[str, str] | None = None,
) -> dict[str, ResourceAttributeValue]:
"""Detect worker-local GPU identity from NVML.

``local_rank`` is the rank's index within ``CUDA_VISIBLE_DEVICES``. Callers
that already know their local rank should pass it explicitly; otherwise the
helper falls back to common launcher environment variables and then rank 0.
"""
env = os.environ if environ is None else environ
physical_index = _physical_gpu_index(local_rank, env)
if physical_index is None:
return {}

try:
import pynvml

pynvml.nvmlInit()
try:
handle = pynvml.nvmlDeviceGetHandleByIndex(physical_index)
attrs: dict[str, ResourceAttributeValue] = {
NV_GPU_INDEX: physical_index,
NV_GPU_MODEL: _decode(pynvml.nvmlDeviceGetName(handle)),
NV_GPU_UUID: _decode(pynvml.nvmlDeviceGetUUID(handle)),
}
_set_optional(attrs, NV_GPU_SERIAL, pynvml.nvmlDeviceGetSerial, handle)
_set_optional_field(
attrs,
NV_GPU_PCI_BUS_ID,
lambda: pynvml.nvmlDeviceGetPciInfo(handle).busId,
)
_set_compute_capability(attrs, pynvml, handle)
_set_optional_field(
attrs,
NV_GPU_MEMORY_TOTAL,
lambda: int(pynvml.nvmlDeviceGetMemoryInfo(handle).total),
)
_set_optional_field(
attrs,
NV_GPU_DRIVER_VERSION,
pynvml.nvmlSystemGetDriverVersion,
)
return attrs
finally:
pynvml.nvmlShutdown()
except Exception:
return {}


def _physical_gpu_index(
local_rank: int | str | None,
env: Mapping[str, str],
) -> int | None:
try:
rank = int(_first_value(local_rank, env, "LOCAL_RANK", "SLURM_LOCALID") or 0)
except (TypeError, ValueError):
rank = 0

cuda_visible = env.get("CUDA_VISIBLE_DEVICES", "").strip()
if not cuda_visible:
return rank

visible = [item.strip() for item in cuda_visible.split(",")]
if rank >= len(visible):
return None
try:
return int(visible[rank])
except ValueError:
return None


def _first_value(
explicit: int | str | None,
env: Mapping[str, str],
*env_names: str,
) -> int | str | None:
if explicit is not None:
return explicit
for name in env_names:
value = env.get(name)
if value not in (None, ""):
return value
return None


def _set_optional(
attrs: dict[str, ResourceAttributeValue],
key: str,
getter,
handle,
) -> None:
_set_optional_field(attrs, key, lambda: getter(handle))


def _set_optional_field(
attrs: dict[str, ResourceAttributeValue],
key: str,
getter,
) -> None:
try:
value = getter()
except Exception:
return
attrs[key] = _decode(value)


def _set_compute_capability(attrs, pynvml, handle) -> None:
try:
major, minor = pynvml.nvmlDeviceGetCudaComputeCapability(handle)
except Exception:
return
attrs[NV_GPU_COMPUTE_CAPABILITY] = f"{major}.{minor}"


def _decode(value):
return value.decode() if isinstance(value, bytes) else value
13 changes: 13 additions & 0 deletions src/nemo/lens/semconv.py
Original file line number Diff line number Diff line change
Expand Up @@ -157,6 +157,19 @@
NV_DL_JOB_UUID = "nv.dl.job.uuid"
NV_DL_RUN_UUID = "nv.dl.run.uuid"

# ------------------------------------------------------------------ #
# GPU identity (nv.gpu.*)
# ------------------------------------------------------------------ #

NV_GPU_UUID = "nv.gpu.uuid"
NV_GPU_INDEX = "nv.gpu.index"
NV_GPU_PCI_BUS_ID = "nv.gpu.pci_bus_id"
NV_GPU_SERIAL = "nv.gpu.serial"
NV_GPU_MODEL = "nv.gpu.model"
NV_GPU_COMPUTE_CAPABILITY = "nv.gpu.compute_capability"
NV_GPU_MEMORY_TOTAL = "nv.gpu.memory_total"
NV_GPU_DRIVER_VERSION = "nv.gpu.driver_version"

# ------------------------------------------------------------------ #
# Run identification (nemo.*)
# ------------------------------------------------------------------ #
Expand Down
150 changes: 150 additions & 0 deletions tests/test_resources.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,11 +17,13 @@

import logging
import os
import sys
import uuid

import pytest

from nemo.lens.resources import (
detect_gpu,
detect_resource,
extend_otel_resource_attributes,
publish_otel_resource_attributes,
Expand Down Expand Up @@ -626,6 +628,154 @@ def test_empty_cuda_visible_means_zero_gpus(self, monkeypatch):
assert result.get("host.gpu.count") == 0


class TestDetectGpu:
def test_detects_worker_gpu_identity(self, monkeypatch):
class FakePciInfo:
busId = b"00000000:81:00.0"

class FakeMemoryInfo:
total = 80_000_000_000

fake_pynvml = type(
"FakePynvml",
(),
{
"nvmlInit": staticmethod(lambda: None),
"nvmlShutdown": staticmethod(lambda: None),
"nvmlDeviceGetHandleByIndex": staticmethod(lambda index: f"gpu-{index}"),
"nvmlDeviceGetName": staticmethod(lambda handle: b"NVIDIA H100"),
"nvmlDeviceGetUUID": staticmethod(lambda handle: b"GPU-123"),
"nvmlDeviceGetSerial": staticmethod(lambda handle: b"serial-123"),
"nvmlDeviceGetPciInfo": staticmethod(lambda handle: FakePciInfo()),
"nvmlDeviceGetCudaComputeCapability": staticmethod(lambda handle: (9, 0)),
"nvmlDeviceGetMemoryInfo": staticmethod(lambda handle: FakeMemoryInfo()),
"nvmlSystemGetDriverVersion": staticmethod(lambda: b"550.54.15"),
},
)
monkeypatch.setitem(sys.modules, "pynvml", fake_pynvml)
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "4,5")

result = detect_gpu(local_rank=1)

assert result == {
"nv.gpu.index": 5,
"nv.gpu.model": "NVIDIA H100",
"nv.gpu.uuid": "GPU-123",
"nv.gpu.serial": "serial-123",
"nv.gpu.pci_bus_id": "00000000:81:00.0",
"nv.gpu.compute_capability": "9.0",
"nv.gpu.memory_total": 80_000_000_000,
"nv.gpu.driver_version": "550.54.15",
}

def test_non_numeric_visible_device_is_not_guessed(self, monkeypatch):
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "GPU-123")

assert detect_gpu(local_rank=0) == {}

def test_pynvml_unavailable_returns_empty(self, monkeypatch):
# Simulate pynvml not installed — import raises ImportError (line 85-86).
monkeypatch.setitem(sys.modules, "pynvml", None)
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0")

assert detect_gpu(local_rank=0) == {}

def test_nvml_init_failure_returns_empty(self, monkeypatch):
# pynvml importable but nvmlInit raises — still caught by outer except (line 85-86).
fake_pynvml = type(
"FakePynvml",
(),
{"nvmlInit": staticmethod(lambda: (_ for _ in ()).throw(RuntimeError("NVML error")))},
)
monkeypatch.setitem(sys.modules, "pynvml", fake_pynvml)
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0")

assert detect_gpu(local_rank=0) == {}

def test_rank_out_of_visible_devices_returns_empty(self, monkeypatch):
# rank >= len(visible) — line 104.
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1")

assert detect_gpu(local_rank=5) == {}

def test_no_cuda_visible_falls_back_to_rank(self, monkeypatch):
# CUDA_VISIBLE_DEVICES absent → line 100 returns rank directly,
# then pynvml is unavailable so the result is {}.
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False)
monkeypatch.setitem(sys.modules, "pynvml", None)

assert detect_gpu(local_rank=0) == {}

def test_local_rank_inferred_from_env(self, monkeypatch):
# local_rank=None → _first_value walks env_names (lines 118-122).
# LOCAL_RANK=0, CUDA_VISIBLE_DEVICES absent, pynvml unavailable → {}.
monkeypatch.setenv("LOCAL_RANK", "0")
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False)
monkeypatch.setitem(sys.modules, "pynvml", None)

assert detect_gpu(local_rank=None) == {}

def test_non_numeric_local_rank_defaults_to_zero(self, monkeypatch):
# LOCAL_RANK is non-numeric → int() raises ValueError → rank = 0 (lines 95-96).
monkeypatch.setenv("LOCAL_RANK", "bad")
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False)
monkeypatch.setitem(sys.modules, "pynvml", None)

assert detect_gpu(local_rank=None) == {}

def test_no_env_rank_defaults_to_zero(self, monkeypatch):
# local_rank=None and all fallback env vars absent → _first_value returns None
# (line 122), int(None) raises TypeError → rank = 0 (lines 95-96).
monkeypatch.delenv("LOCAL_RANK", raising=False)
monkeypatch.delenv("SLURM_LOCALID", raising=False)
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False)
monkeypatch.setitem(sys.modules, "pynvml", None)

assert detect_gpu(local_rank=None) == {}

def test_optional_field_exception_is_silenced(self, monkeypatch):
# A getter in _set_optional_field raises — attribute is skipped (lines 141-142).
fake_pynvml = type(
"FakePynvml",
(),
{
"nvmlInit": staticmethod(lambda: None),
"nvmlShutdown": staticmethod(lambda: None),
"nvmlDeviceGetHandleByIndex": staticmethod(lambda index: "gpu-0"),
"nvmlDeviceGetName": staticmethod(lambda handle: b"NVIDIA H100"),
"nvmlDeviceGetUUID": staticmethod(lambda handle: b"GPU-abc"),
# Serial raises → skipped silently.
"nvmlDeviceGetSerial": staticmethod(
lambda handle: (_ for _ in ()).throw(RuntimeError("not supported"))
),
# PCI info raises too.
"nvmlDeviceGetPciInfo": staticmethod(
lambda handle: (_ for _ in ()).throw(RuntimeError("not supported"))
),
# Compute capability raises → lines 149-150.
"nvmlDeviceGetCudaComputeCapability": staticmethod(
lambda handle: (_ for _ in ()).throw(RuntimeError("not supported"))
),
# Memory info raises.
"nvmlDeviceGetMemoryInfo": staticmethod(
lambda handle: (_ for _ in ()).throw(RuntimeError("not supported"))
),
"nvmlSystemGetDriverVersion": staticmethod(lambda: b"550.0"),
},
)
monkeypatch.setitem(sys.modules, "pynvml", fake_pynvml)
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0")

result = detect_gpu(local_rank=0)

assert result["nv.gpu.model"] == "NVIDIA H100"
assert result["nv.gpu.uuid"] == "GPU-abc"
assert "nv.gpu.serial" not in result
assert "nv.gpu.pci_bus_id" not in result
assert "nv.gpu.compute_capability" not in result
assert "nv.gpu.memory_total" not in result


class TestDetectResource:
def test_always_returns_dict(self, monkeypatch):
monkeypatch.delenv("SLURM_JOB_ID", raising=False)
Expand Down
Loading