diff --git a/src/nemo/lens/resources/__init__.py b/src/nemo/lens/resources/__init__.py index d7b78a5..8c3db68 100644 --- a/src/nemo/lens/resources/__init__.py +++ b/src/nemo/lens/resources/__init__.py @@ -20,6 +20,7 @@ publish_otel_resource_attributes, set_otel_resource_attributes, ) +from nemo.lens.resources.gpu import detect_gpu from nemo.lens.resources.kubernetes import detect_kubernetes from nemo.lens.resources.local import detect_local from nemo.lens.resources.slurm import detect_slurm @@ -46,4 +47,5 @@ def detect_resource() -> dict: "extend_otel_resource_attributes", "set_otel_resource_attributes", "publish_otel_resource_attributes", + "detect_gpu", ] diff --git a/src/nemo/lens/resources/gpu.py b/src/nemo/lens/resources/gpu.py new file mode 100644 index 0000000..9aae140 --- /dev/null +++ b/src/nemo/lens/resources/gpu.py @@ -0,0 +1,155 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""GPU identity detection.""" + +from __future__ import annotations + +import os +from collections.abc import Mapping +from typing import TYPE_CHECKING + +from nemo.lens.semconv import ( + NV_GPU_COMPUTE_CAPABILITY, + NV_GPU_DRIVER_VERSION, + NV_GPU_INDEX, + NV_GPU_MEMORY_TOTAL, + NV_GPU_MODEL, + NV_GPU_PCI_BUS_ID, + NV_GPU_SERIAL, + NV_GPU_UUID, +) + +if TYPE_CHECKING: + from nemo.lens.resources.attributes import ResourceAttributeValue + + +def detect_gpu( + local_rank: int | str | None = None, + environ: Mapping[str, str] | None = None, +) -> dict[str, ResourceAttributeValue]: + """Detect worker-local GPU identity from NVML. + + ``local_rank`` is the rank's index within ``CUDA_VISIBLE_DEVICES``. Callers + that already know their local rank should pass it explicitly; otherwise the + helper falls back to common launcher environment variables and then rank 0. + """ + env = os.environ if environ is None else environ + physical_index = _physical_gpu_index(local_rank, env) + if physical_index is None: + return {} + + try: + import pynvml + + pynvml.nvmlInit() + try: + handle = pynvml.nvmlDeviceGetHandleByIndex(physical_index) + attrs: dict[str, ResourceAttributeValue] = { + NV_GPU_INDEX: physical_index, + NV_GPU_MODEL: _decode(pynvml.nvmlDeviceGetName(handle)), + NV_GPU_UUID: _decode(pynvml.nvmlDeviceGetUUID(handle)), + } + _set_optional(attrs, NV_GPU_SERIAL, pynvml.nvmlDeviceGetSerial, handle) + _set_optional_field( + attrs, + NV_GPU_PCI_BUS_ID, + lambda: pynvml.nvmlDeviceGetPciInfo(handle).busId, + ) + _set_compute_capability(attrs, pynvml, handle) + _set_optional_field( + attrs, + NV_GPU_MEMORY_TOTAL, + lambda: int(pynvml.nvmlDeviceGetMemoryInfo(handle).total), + ) + _set_optional_field( + attrs, + NV_GPU_DRIVER_VERSION, + pynvml.nvmlSystemGetDriverVersion, + ) + return attrs + finally: + pynvml.nvmlShutdown() + except Exception: + return {} + + +def _physical_gpu_index( + local_rank: int | str | None, + env: Mapping[str, str], +) -> int | None: + try: + rank = int(_first_value(local_rank, env, "LOCAL_RANK", "SLURM_LOCALID") or 0) + except (TypeError, ValueError): + rank = 0 + + cuda_visible = env.get("CUDA_VISIBLE_DEVICES", "").strip() + if not cuda_visible: + return rank + + visible = [item.strip() for item in cuda_visible.split(",")] + if rank >= len(visible): + return None + try: + return int(visible[rank]) + except ValueError: + return None + + +def _first_value( + explicit: int | str | None, + env: Mapping[str, str], + *env_names: str, +) -> int | str | None: + if explicit is not None: + return explicit + for name in env_names: + value = env.get(name) + if value not in (None, ""): + return value + return None + + +def _set_optional( + attrs: dict[str, ResourceAttributeValue], + key: str, + getter, + handle, +) -> None: + _set_optional_field(attrs, key, lambda: getter(handle)) + + +def _set_optional_field( + attrs: dict[str, ResourceAttributeValue], + key: str, + getter, +) -> None: + try: + value = getter() + except Exception: + return + attrs[key] = _decode(value) + + +def _set_compute_capability(attrs, pynvml, handle) -> None: + try: + major, minor = pynvml.nvmlDeviceGetCudaComputeCapability(handle) + except Exception: + return + attrs[NV_GPU_COMPUTE_CAPABILITY] = f"{major}.{minor}" + + +def _decode(value): + return value.decode() if isinstance(value, bytes) else value diff --git a/src/nemo/lens/semconv.py b/src/nemo/lens/semconv.py index 48370d6..aeb4efa 100644 --- a/src/nemo/lens/semconv.py +++ b/src/nemo/lens/semconv.py @@ -157,6 +157,19 @@ NV_DL_JOB_UUID = "nv.dl.job.uuid" NV_DL_RUN_UUID = "nv.dl.run.uuid" +# ------------------------------------------------------------------ # +# GPU identity (nv.gpu.*) +# ------------------------------------------------------------------ # + +NV_GPU_UUID = "nv.gpu.uuid" +NV_GPU_INDEX = "nv.gpu.index" +NV_GPU_PCI_BUS_ID = "nv.gpu.pci_bus_id" +NV_GPU_SERIAL = "nv.gpu.serial" +NV_GPU_MODEL = "nv.gpu.model" +NV_GPU_COMPUTE_CAPABILITY = "nv.gpu.compute_capability" +NV_GPU_MEMORY_TOTAL = "nv.gpu.memory_total" +NV_GPU_DRIVER_VERSION = "nv.gpu.driver_version" + # ------------------------------------------------------------------ # # Run identification (nemo.*) # ------------------------------------------------------------------ # diff --git a/tests/test_resources.py b/tests/test_resources.py index f348441..b4bd4d8 100644 --- a/tests/test_resources.py +++ b/tests/test_resources.py @@ -17,11 +17,13 @@ import logging import os +import sys import uuid import pytest from nemo.lens.resources import ( + detect_gpu, detect_resource, extend_otel_resource_attributes, publish_otel_resource_attributes, @@ -626,6 +628,154 @@ def test_empty_cuda_visible_means_zero_gpus(self, monkeypatch): assert result.get("host.gpu.count") == 0 +class TestDetectGpu: + def test_detects_worker_gpu_identity(self, monkeypatch): + class FakePciInfo: + busId = b"00000000:81:00.0" + + class FakeMemoryInfo: + total = 80_000_000_000 + + fake_pynvml = type( + "FakePynvml", + (), + { + "nvmlInit": staticmethod(lambda: None), + "nvmlShutdown": staticmethod(lambda: None), + "nvmlDeviceGetHandleByIndex": staticmethod(lambda index: f"gpu-{index}"), + "nvmlDeviceGetName": staticmethod(lambda handle: b"NVIDIA H100"), + "nvmlDeviceGetUUID": staticmethod(lambda handle: b"GPU-123"), + "nvmlDeviceGetSerial": staticmethod(lambda handle: b"serial-123"), + "nvmlDeviceGetPciInfo": staticmethod(lambda handle: FakePciInfo()), + "nvmlDeviceGetCudaComputeCapability": staticmethod(lambda handle: (9, 0)), + "nvmlDeviceGetMemoryInfo": staticmethod(lambda handle: FakeMemoryInfo()), + "nvmlSystemGetDriverVersion": staticmethod(lambda: b"550.54.15"), + }, + ) + monkeypatch.setitem(sys.modules, "pynvml", fake_pynvml) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "4,5") + + result = detect_gpu(local_rank=1) + + assert result == { + "nv.gpu.index": 5, + "nv.gpu.model": "NVIDIA H100", + "nv.gpu.uuid": "GPU-123", + "nv.gpu.serial": "serial-123", + "nv.gpu.pci_bus_id": "00000000:81:00.0", + "nv.gpu.compute_capability": "9.0", + "nv.gpu.memory_total": 80_000_000_000, + "nv.gpu.driver_version": "550.54.15", + } + + def test_non_numeric_visible_device_is_not_guessed(self, monkeypatch): + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "GPU-123") + + assert detect_gpu(local_rank=0) == {} + + def test_pynvml_unavailable_returns_empty(self, monkeypatch): + # Simulate pynvml not installed — import raises ImportError (line 85-86). + monkeypatch.setitem(sys.modules, "pynvml", None) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") + + assert detect_gpu(local_rank=0) == {} + + def test_nvml_init_failure_returns_empty(self, monkeypatch): + # pynvml importable but nvmlInit raises — still caught by outer except (line 85-86). + fake_pynvml = type( + "FakePynvml", + (), + {"nvmlInit": staticmethod(lambda: (_ for _ in ()).throw(RuntimeError("NVML error")))}, + ) + monkeypatch.setitem(sys.modules, "pynvml", fake_pynvml) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") + + assert detect_gpu(local_rank=0) == {} + + def test_rank_out_of_visible_devices_returns_empty(self, monkeypatch): + # rank >= len(visible) — line 104. + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1") + + assert detect_gpu(local_rank=5) == {} + + def test_no_cuda_visible_falls_back_to_rank(self, monkeypatch): + # CUDA_VISIBLE_DEVICES absent → line 100 returns rank directly, + # then pynvml is unavailable so the result is {}. + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + monkeypatch.setitem(sys.modules, "pynvml", None) + + assert detect_gpu(local_rank=0) == {} + + def test_local_rank_inferred_from_env(self, monkeypatch): + # local_rank=None → _first_value walks env_names (lines 118-122). + # LOCAL_RANK=0, CUDA_VISIBLE_DEVICES absent, pynvml unavailable → {}. + monkeypatch.setenv("LOCAL_RANK", "0") + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + monkeypatch.setitem(sys.modules, "pynvml", None) + + assert detect_gpu(local_rank=None) == {} + + def test_non_numeric_local_rank_defaults_to_zero(self, monkeypatch): + # LOCAL_RANK is non-numeric → int() raises ValueError → rank = 0 (lines 95-96). + monkeypatch.setenv("LOCAL_RANK", "bad") + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + monkeypatch.setitem(sys.modules, "pynvml", None) + + assert detect_gpu(local_rank=None) == {} + + def test_no_env_rank_defaults_to_zero(self, monkeypatch): + # local_rank=None and all fallback env vars absent → _first_value returns None + # (line 122), int(None) raises TypeError → rank = 0 (lines 95-96). + monkeypatch.delenv("LOCAL_RANK", raising=False) + monkeypatch.delenv("SLURM_LOCALID", raising=False) + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + monkeypatch.setitem(sys.modules, "pynvml", None) + + assert detect_gpu(local_rank=None) == {} + + def test_optional_field_exception_is_silenced(self, monkeypatch): + # A getter in _set_optional_field raises — attribute is skipped (lines 141-142). + fake_pynvml = type( + "FakePynvml", + (), + { + "nvmlInit": staticmethod(lambda: None), + "nvmlShutdown": staticmethod(lambda: None), + "nvmlDeviceGetHandleByIndex": staticmethod(lambda index: "gpu-0"), + "nvmlDeviceGetName": staticmethod(lambda handle: b"NVIDIA H100"), + "nvmlDeviceGetUUID": staticmethod(lambda handle: b"GPU-abc"), + # Serial raises → skipped silently. + "nvmlDeviceGetSerial": staticmethod( + lambda handle: (_ for _ in ()).throw(RuntimeError("not supported")) + ), + # PCI info raises too. + "nvmlDeviceGetPciInfo": staticmethod( + lambda handle: (_ for _ in ()).throw(RuntimeError("not supported")) + ), + # Compute capability raises → lines 149-150. + "nvmlDeviceGetCudaComputeCapability": staticmethod( + lambda handle: (_ for _ in ()).throw(RuntimeError("not supported")) + ), + # Memory info raises. + "nvmlDeviceGetMemoryInfo": staticmethod( + lambda handle: (_ for _ in ()).throw(RuntimeError("not supported")) + ), + "nvmlSystemGetDriverVersion": staticmethod(lambda: b"550.0"), + }, + ) + monkeypatch.setitem(sys.modules, "pynvml", fake_pynvml) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") + + result = detect_gpu(local_rank=0) + + assert result["nv.gpu.model"] == "NVIDIA H100" + assert result["nv.gpu.uuid"] == "GPU-abc" + assert "nv.gpu.serial" not in result + assert "nv.gpu.pci_bus_id" not in result + assert "nv.gpu.compute_capability" not in result + assert "nv.gpu.memory_total" not in result + + class TestDetectResource: def test_always_returns_dict(self, monkeypatch): monkeypatch.delenv("SLURM_JOB_ID", raising=False)