From 76eee5cf9a77b3d58d3d7304f7c2d95d1ac3a2f9 Mon Sep 17 00:00:00 2001 From: li-lizhe <147392333@qq.com> Date: Sat, 3 Oct 2026 11:18:41 +0800 Subject: [PATCH] Rebase onto main@c4d63ef207 Resolved conflicts with `git merge-file`; change set unchanged. Signed-off-by: li-lizhe <147392333@qq.com> --- CHANGELOG.md | 1 + backend/api/routers/system.py | 50 +++++--- .../backend/api/test_system_gpu_detection.py | 112 ++++++++++++++++++ 3 files changed, 148 insertions(+), 15 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a092ccf22..eebcaf420 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -101,6 +101,7 @@ metadata and the backend fallback mirror it. - License notice: commercial use is free under the AGPL; the paid licence is for closed-source use, with Pro plans linked (#2578) ### Fixed +- Settings → System reports an Ascend NPU host's device name and VRAM, and the memory-flush snapshot reads the NPU and Intel XPU (#2582) - Elevated Windows app removal stops before deleting data and points to a normal PowerShell window or Settings (#2578) - Contributor audits inspect committed files and exclude submodules, while still stopping on failed file attribution (#2556) diff --git a/backend/api/routers/system.py b/backend/api/routers/system.py index 165b493dd..33c7a5362 100644 --- a/backend/api/routers/system.py +++ b/backend/api/routers/system.py @@ -45,6 +45,10 @@ _is_xpu = hasattr(torch, "xpu") and torch.xpu.is_available() except Exception: _is_xpu = False +try: + _is_npu = hasattr(torch, "npu") and torch.npu.is_available() +except Exception: + _is_npu = False # Prime psutil's internal CPU counter so the first non-blocking call returns useful data psutil.cpu_percent(interval=None) @@ -147,20 +151,37 @@ def _detect_os_gpu_name() -> str: return "" +def _active_accelerator(): + """(module, name) of the accelerator this host runs on, or (None, ""). + + One resolution shared by device detection, ``/sysinfo`` and the post-flush + snapshot, so a host's backend is decided in a single place. CUDA, Intel XPU + and Ascend NPU all expose ``get_device_name`` / ``get_device_properties`` / + ``memory_allocated`` / ``memory_reserved`` through the same shape, so the + same code covers them. MPS is unified-memory and keeps its own branch (it + reports no device-side total). + """ + if _is_cuda: + return torch.cuda, "cuda" + if _is_xpu: + return torch.xpu, "xpu" + if _is_npu: + return torch.npu, "npu" + return None, "" + + def _detect_gpu() -> tuple[str, float]: """(gpu_name, vram_total_gb) — static for the process lifetime. MPS has unified memory, so there's no separate VRAM figure to report; the name alone tells a bug-report reader what hardware this is. """ + backend, _ = _active_accelerator() try: - if _is_cuda: - props = torch.cuda.get_device_properties(0) - return torch.cuda.get_device_name(0), round(props.total_memory / (1024 ** 3), 1) - if _is_xpu: - props = torch.xpu.get_device_properties(0) + if backend is not None: + props = backend.get_device_properties(0) total_memory = float(getattr(props, "total_memory", 0.0)) - return torch.xpu.get_device_name(0), round(total_memory / (1024 ** 3), 1) + return backend.get_device_name(0), round(total_memory / (1024 ** 3), 1) if _is_mac: return "Apple Silicon (MPS)", 0.0 except Exception: @@ -807,6 +828,7 @@ def get_sys_info(): total_vram = 0.0 gpu_active = False + backend, _ = _active_accelerator() try: if _is_mac: alloc = getattr(torch.mps, "current_allocated_memory", None) @@ -815,13 +837,10 @@ def get_sys_info(): vram = driver() / (1024**3) elif alloc: vram = alloc() / (1024**3) - elif _is_cuda: - vram = torch.cuda.memory_allocated() / (1024**3) - total_vram = torch.cuda.get_device_properties(torch.cuda.current_device()).total_memory / (1024**3) - elif _is_xpu: - vram = torch.xpu.memory_allocated() / (1024**3) + elif backend is not None: + vram = backend.memory_allocated() / (1024**3) total_vram = float( - getattr(torch.xpu.get_device_properties(0), "total_memory", 0.0) + getattr(backend.get_device_properties(backend.current_device()), "total_memory", 0.0) ) / (1024**3) except Exception: pass @@ -893,6 +912,7 @@ async def flush_memory(unload_model: bool = False): # CUDA context plus kernel workspaces, which no in-process call can return. vram_after = 0.0 vram_reserved = 0.0 + backend, _ = _active_accelerator() try: if hasattr(torch.backends, "mps") and torch.backends.mps.is_available(): driver = getattr(torch.mps, "driver_allocated_memory", None) @@ -900,9 +920,9 @@ async def flush_memory(unload_model: bool = False): vram_after = driver() / (1024**3) current = getattr(torch.mps, "current_allocated_memory", None) vram_reserved = (current() / (1024**3)) if current else vram_after - elif torch.cuda.is_available(): - vram_after = torch.cuda.memory_allocated() / (1024**3) - vram_reserved = torch.cuda.memory_reserved() / (1024**3) + elif backend is not None: + vram_after = backend.memory_allocated() / (1024**3) + vram_reserved = backend.memory_reserved() / (1024**3) except Exception: pass diff --git a/tests/backend/api/test_system_gpu_detection.py b/tests/backend/api/test_system_gpu_detection.py index f09bb6528..9d438e6af 100644 --- a/tests/backend/api/test_system_gpu_detection.py +++ b/tests/backend/api/test_system_gpu_detection.py @@ -32,3 +32,115 @@ def test_linux_gpu_fallback_reads_lspci_machine_output(monkeypatch): ) assert system._detect_os_gpu_name() == "NVIDIA Corporation GeForce RTX 4090" + + +def _npu_backend(*, name="Ascend 910B", total_gb=64.0, allocated_gb=1.5, reserved_gb=2.0): + """A torch.npu-shaped stub: the surface ``_active_accelerator`` callers use.""" + return SimpleNamespace( + is_available=lambda: True, + get_device_name=lambda index: name, + get_device_properties=lambda index: SimpleNamespace( + total_memory=int(total_gb * 1024 ** 3) + ), + current_device=lambda: 0, + memory_allocated=lambda: int(allocated_gb * 1024 ** 3), + memory_reserved=lambda: int(reserved_gb * 1024 ** 3), + ) + + +def _torch_stub(**accelerators): + """A torch-module stub carrying the attributes the router touches. + + ``backends`` is always present (like the real torch) so the MPS probe in + ``flush_memory`` reads False instead of raising on attribute access. + """ + return SimpleNamespace(backends=SimpleNamespace(), **accelerators) + + +def _cpu_only_flags(monkeypatch): + monkeypatch.setattr(system, "_is_mac", False) + monkeypatch.setattr(system, "_is_cuda", False) + monkeypatch.setattr(system, "_is_xpu", False) + + +def test_detect_gpu_reports_ascend_npu(monkeypatch): + _cpu_only_flags(monkeypatch) + monkeypatch.setattr(system, "_is_npu", True) + monkeypatch.setattr(system, "torch", _torch_stub(npu=_npu_backend())) + + assert system._detect_gpu() == ("Ascend 910B", 64.0) + + +def test_detect_gpu_prefers_cuda_over_npu(monkeypatch): + monkeypatch.setattr(system, "_is_mac", False) + monkeypatch.setattr(system, "_is_cuda", True) + monkeypatch.setattr(system, "_is_xpu", False) + monkeypatch.setattr(system, "_is_npu", True, raising=False) + cuda = SimpleNamespace( + get_device_name=lambda index: "NVIDIA RTX 4090", + get_device_properties=lambda index: SimpleNamespace(total_memory=24 * 1024 ** 3), + ) + monkeypatch.setattr(system, "torch", _torch_stub(cuda=cuda, npu=_npu_backend())) + + assert system._detect_gpu() == ("NVIDIA RTX 4090", 24.0) + + +def test_detect_gpu_cpu_host_still_falls_back_to_os_probe(monkeypatch): + _cpu_only_flags(monkeypatch) + monkeypatch.setattr(system, "_is_npu", False, raising=False) + monkeypatch.setattr(system, "torch", _torch_stub()) + monkeypatch.setattr(system, "_detect_os_gpu_name", lambda: "Intel UHD Graphics 770") + + assert system._detect_gpu() == ("Intel UHD Graphics 770", 0.0) + + +def test_sysinfo_reports_npu_memory(monkeypatch): + _cpu_only_flags(monkeypatch) + monkeypatch.setattr(system, "_is_npu", True) + monkeypatch.setattr(system, "_GPU_NAME", "Ascend 910B") + monkeypatch.setattr( + system, "torch", _torch_stub(npu=_npu_backend(total_gb=64.0, allocated_gb=3.0)) + ) + monkeypatch.setattr( + system, + "psutil", + SimpleNamespace( + cpu_percent=lambda interval=None: 12.5, + cpu_count=lambda logical=True: 8, + cpu_freq=lambda: SimpleNamespace(current=2400.0), + virtual_memory=lambda: SimpleNamespace(used=8 * 1024 ** 3, total=32 * 1024 ** 3), + ), + ) + + info = system.get_sys_info() + + assert info["vram"] == 3.0 + assert info["total_vram"] == 64.0 + assert info["gpu_active"] is True + + +def test_flush_memory_snapshot_reads_npu(monkeypatch): + import asyncio + + from services import model_manager + + _cpu_only_flags(monkeypatch) + monkeypatch.setattr(system, "_is_npu", True) + monkeypatch.setattr( + system, + "torch", + _torch_stub(npu=_npu_backend(allocated_gb=0.5, reserved_gb=1.25)), + ) + monkeypatch.setattr(model_manager, "free_vram", lambda: None) + monkeypatch.setattr( + system, + "psutil", + SimpleNamespace( + virtual_memory=lambda: SimpleNamespace(used=8 * 1024 ** 3, total=32 * 1024 ** 3) + ), + ) + + result = asyncio.run(system.flush_memory(unload_model=False)) + + assert result["vram_after"] == 0.5 + assert result["vram_reserved"] == 1.25