-
-
Notifications
You must be signed in to change notification settings - Fork 5.8k
fix(system): report a host Ascend NPU in device info and the flush snapshot #2582
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
base: main
Are you sure you want to change the base?
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -45,6 +45,10 @@ | |
| _is_xpu = hasattr(torch, "xpu") and torch.xpu.is_available() | ||
| except Exception: | ||
| _is_xpu = False | ||
| try: | ||
| _is_npu = hasattr(torch, "npu") and torch.npu.is_available() | ||
| except Exception: | ||
| _is_npu = False | ||
| # Prime psutil's internal CPU counter so the first non-blocking call returns useful data | ||
| psutil.cpu_percent(interval=None) | ||
|
|
||
|
|
@@ -147,20 +151,37 @@ def _detect_os_gpu_name() -> str: | |
| return "" | ||
|
|
||
|
|
||
| def _active_accelerator(): | ||
| """(module, name) of the accelerator this host runs on, or (None, ""). | ||
|
|
||
| One resolution shared by device detection, ``/sysinfo`` and the post-flush | ||
| snapshot, so a host's backend is decided in a single place. CUDA, Intel XPU | ||
| and Ascend NPU all expose ``get_device_name`` / ``get_device_properties`` / | ||
| ``memory_allocated`` / ``memory_reserved`` through the same shape, so the | ||
| same code covers them. MPS is unified-memory and keeps its own branch (it | ||
| reports no device-side total). | ||
| """ | ||
| if _is_cuda: | ||
| return torch.cuda, "cuda" | ||
| if _is_xpu: | ||
| return torch.xpu, "xpu" | ||
| if _is_npu: | ||
| return torch.npu, "npu" | ||
| return None, "" | ||
|
|
||
|
|
||
| def _detect_gpu() -> tuple[str, float]: | ||
| """(gpu_name, vram_total_gb) — static for the process lifetime. | ||
|
|
||
| MPS has unified memory, so there's no separate VRAM figure to report; | ||
| the name alone tells a bug-report reader what hardware this is. | ||
| """ | ||
| backend, _ = _active_accelerator() | ||
| try: | ||
| if _is_cuda: | ||
| props = torch.cuda.get_device_properties(0) | ||
| return torch.cuda.get_device_name(0), round(props.total_memory / (1024 ** 3), 1) | ||
| if _is_xpu: | ||
| props = torch.xpu.get_device_properties(0) | ||
| if backend is not None: | ||
| props = backend.get_device_properties(0) | ||
| total_memory = float(getattr(props, "total_memory", 0.0)) | ||
| return torch.xpu.get_device_name(0), round(total_memory / (1024 ** 3), 1) | ||
| return backend.get_device_name(0), round(total_memory / (1024 ** 3), 1) | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🎯 Functional Correctness | 🟡 Minor | ⚡ Quick win 🔎 Supported by static analysis🏁 Script executed: git diff befc0a6f5b552b0b99e6eea574bc3c55e3d0519a 01a6f79d7bcdfd8fc0d52b8cfcf0aca880e93d1b -- backend/api/routers/system.py
sed -n '140,205p' backend/api/routers/system.py
sed -n '795,860p' backend/api/routers/system.py
sed -n '895,940p' backend/api/routers/system.py
rg -n '_detect_gpu|gpu_total_memory|gpu_name|current_device' backend/api/routers/system.py tests/backend/api/test_system_gpu_detection.pyRepository: debpalash/VoiceStudio Length of output: 13420 🏁 Script executed: sed -n '1,125p' tests/backend/api/test_system_gpu_detection.py
sed -n '1,75p' backend/api/routers/system.py
rg -n -C 3 'current_device|set_device|device\(0\)|get_device_properties|get_device_name|memory_allocated|memory_reserved|_GPU_NAME|_VRAM_TOTAL_GB' backend/api/routers/system.py tests/backend/api
git diff --stat befc0a6f5b552b0b99e6eea574bc3c55e3d0519a 01a6f79d7bcdfd8fc0d52b8cfcf0aca880e93d1bRepository: debpalash/VoiceStudio Length of output: 16618 Use device 0 consistently for NPU memory statistics.
🤖 Prompt for AI Agents |
||
| if _is_mac: | ||
| return "Apple Silicon (MPS)", 0.0 | ||
| except Exception: | ||
|
|
@@ -807,6 +828,7 @@ def get_sys_info(): | |
| total_vram = 0.0 | ||
| gpu_active = False | ||
|
|
||
| backend, _ = _active_accelerator() | ||
| try: | ||
| if _is_mac: | ||
| alloc = getattr(torch.mps, "current_allocated_memory", None) | ||
|
|
@@ -815,13 +837,10 @@ def get_sys_info(): | |
| vram = driver() / (1024**3) | ||
| elif alloc: | ||
| vram = alloc() / (1024**3) | ||
| elif _is_cuda: | ||
| vram = torch.cuda.memory_allocated() / (1024**3) | ||
| total_vram = torch.cuda.get_device_properties(torch.cuda.current_device()).total_memory / (1024**3) | ||
| elif _is_xpu: | ||
| vram = torch.xpu.memory_allocated() / (1024**3) | ||
| elif backend is not None: | ||
| vram = backend.memory_allocated() / (1024**3) | ||
| total_vram = float( | ||
| getattr(torch.xpu.get_device_properties(0), "total_memory", 0.0) | ||
| getattr(backend.get_device_properties(backend.current_device()), "total_memory", 0.0) | ||
| ) / (1024**3) | ||
| except Exception: | ||
| pass | ||
|
|
@@ -893,16 +912,17 @@ async def flush_memory(unload_model: bool = False): | |
| # CUDA context plus kernel workspaces, which no in-process call can return. | ||
| vram_after = 0.0 | ||
| vram_reserved = 0.0 | ||
| backend, _ = _active_accelerator() | ||
| try: | ||
| if hasattr(torch.backends, "mps") and torch.backends.mps.is_available(): | ||
| driver = getattr(torch.mps, "driver_allocated_memory", None) | ||
| if driver: | ||
| vram_after = driver() / (1024**3) | ||
| current = getattr(torch.mps, "current_allocated_memory", None) | ||
| vram_reserved = (current() / (1024**3)) if current else vram_after | ||
| elif torch.cuda.is_available(): | ||
| vram_after = torch.cuda.memory_allocated() / (1024**3) | ||
| vram_reserved = torch.cuda.memory_reserved() / (1024**3) | ||
| elif backend is not None: | ||
| vram_after = backend.memory_allocated() / (1024**3) | ||
| vram_reserved = backend.memory_reserved() / (1024**3) | ||
| except Exception: | ||
| pass | ||
|
|
||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
🎯 Functional Correctness | 🟡 Minor | ⚡ Quick win
🔎 Supported by static analysis
🏁 Script executed:
Repository: debpalash/VoiceStudio
Length of output: 41778
🤖 get_repo_knowledge executed:
get_repo_knowledge debpalash/VoiceStudio /tmp/coderabbit-repo-knowledge/debpalash-voicestudio-90b9abec/conventions /tmp/coderabbit-repo-knowledge/debpalash-voicestudio-90b9abec/learningsLength of output: 13886
🏁 Script executed:
Repository: debpalash/VoiceStudio
Length of output: 41447
🏁 Script executed:
Repository: debpalash/VoiceStudio
Length of output: 45283
🏁 Script executed:
Repository: debpalash/VoiceStudio
Length of output: 25194
🌐 Web query:
official PyTorch extension autoloading torch 2.5 torch_npu documentation💡 Result:
Load
torch_npubefore the cached probe.backend/api/routers/system.py:49caches NPU availability before the router fan-out completes, and the project supportstorch>=2.4without an explicittorch_npuimport; on installations without extension autoloading, all three reporting paths omit NPU data. Importtorch_npubefore this router loads, or probe NPU after registration.🤖 Prompt for AI Agents