test/app/core/accelerators/nvidia.py

86 lines
2.7 KiB
Python

# -*- coding: utf-8 -*-
"""NVIDIA CUDA accelerator adapter."""
from __future__ import annotations
from .base import (
AcceleratorInfo,
command_exists,
first_env_devices,
parse_memory_gb,
run_command,
)
class NvidiaAcceleratorAdapter:
vendor = "nvidia"
runtime = "cuda"
smi_command = "nvidia-smi"
def detect(self) -> AcceleratorInfo:
visible_devices = first_env_devices(("CUDA_VISIBLE_DEVICES",))
smi_available = command_exists(self.smi_command)
smi_count = 0
smi_memory_gb = 0.0
if smi_available:
indexes_output = run_command(
[
self.smi_command,
"--query-gpu=index",
"--format=csv,noheader,nounits",
]
)
if indexes_output:
smi_count = len([line for line in indexes_output.splitlines() if line.strip()])
memory_output = run_command(
[
self.smi_command,
"--query-gpu=memory.total",
"--format=csv,noheader",
]
)
smi_memory_gb = parse_memory_gb(memory_output)
torch_available = False
torch_count = 0
torch_memory_gb = 0.0
torch_version = ""
try:
import torch
torch_version = getattr(torch, "__version__", "")
torch_available = bool(torch.cuda.is_available())
if torch_available:
torch_count = int(torch.cuda.device_count())
torch_memory_gb = min(
torch.cuda.get_device_properties(i).total_memory / (1024**3)
for i in range(torch_count)
)
except Exception:
pass
device_count = torch_count or smi_count
if visible_devices:
device_count = min(device_count or len(visible_devices), len(visible_devices))
total_memory_gb = torch_memory_gb or smi_memory_gb
available = bool(torch_available or smi_count > 0)
reason = "" if available else "nvidia runtime not detected"
return AcceleratorInfo(
vendor=self.vendor,
runtime=self.runtime,
device="cuda:0" if available else "cpu",
device_count=device_count,
visible_devices=visible_devices,
total_memory_gb=total_memory_gb,
smi_command=self.smi_command if smi_available else None,
available=available,
reason=reason,
metadata={
"torch_cuda_available": torch_available,
"torch_version": torch_version,
"supports_sharded": available,
},
)