From 6757ca9e61d9e12173d1c3b8ba7a2d52cf29d649 Mon Sep 17 00:00:00 2001 From: Meisterlala <6453306+Meisterlala@users.noreply.github.com> Date: Mon, 3 Aug 2026 17:43:52 +0200 Subject: [PATCH] feat(llamaModels): enhance hover view to display detailed resource usage metrics added VRAM, parameter count, and context size to the model hover view for improved user insights about resource utilization. removed specific RAM and offload estimates to avoid misleading information on memory distribution. --- main/modules/LlamaModels.qml | 36 +++++++++---------- main/scripts/llama_models.py | 67 +++++++++++++++++++++++++++++++++--- 2 files changed, 79 insertions(+), 24 deletions(-) diff --git a/main/modules/LlamaModels.qml b/main/modules/LlamaModels.qml index d07295b..7dd99bd 100644 --- a/main/modules/LlamaModels.qml +++ b/main/modules/LlamaModels.qml @@ -67,7 +67,7 @@ ClippingRectangle { let state = generationSamples[model.id]; if (!state || state.generationId !== model.generationId) - state = { generationId: model.generationId, samples: [], phase: "" }; + state = { generationId: model.generationId, samples: [], phase: "prompt" }; const sample = { timestamp: now, @@ -89,12 +89,10 @@ ClippingRectangle { const generatedTokens = Math.max(0, sample.generated - first.generated); if (generatedTokens > 0) state.phase = "generation"; - else if (promptTokens > 0 || (state.phase.length === 0 && sample.prompt > 0)) - state.phase = "prompt"; const tokens = state.phase === "prompt" ? promptTokens : generatedTokens; model.tokensPerSecond = elapsed > 0 ? Math.max(0, tokens) * 1000 / elapsed : 0; - model.ratePhase = state.phase || "generation"; + model.ratePhase = state.phase; } function rateText(value) { @@ -131,26 +129,25 @@ ClippingRectangle { function paramsText(value) { const params = Number(value || 0); - return params > 0 ? `${(params / 1000000000).toFixed(1)}B params` : ""; + return params > 0 ? `${(params / 1000000000).toFixed(1)}B param` : ""; } - function sizeText(value) { - const size = Number(value || 0); - return size > 0 ? `${(size / 1073741824).toFixed(1)} GiB` : ""; + function contextText(value) { + const context = Number(value || 0); + return context > 0 ? `${Math.round(context / 1024)}k ctx` : ""; } - function modelDetails(model) { + function modelSpec(model) { const details = []; + const vram = Number(model.vramBytes || 0); const params = paramsText(model.params); - const size = sizeText(model.size); + const context = contextText(model.contextSize); + if (vram > 0) + details.push(`${(vram / 1073741824).toFixed(1)} GiB VRAM`); if (params.length > 0) details.push(params); - if (size.length > 0) - details.push(size); - if (Number(model.contextSize || 0) > 0) - details.push(`${Number(model.contextSize).toLocaleString()} ctx`); - if (Array.isArray(model.modalities) && model.modalities.length > 0) - details.push(model.modalities.join(" + ")); + if (context.length > 0) + details.push(context); return details.join(" ยท "); } @@ -304,9 +301,10 @@ ClippingRectangle { Row { width: parent.width + spacing: 8 Text { - width: parent.width - stateLabel.width + width: parent.width - stateLabel.width - parent.spacing color: theme.text elide: Text.ElideMiddle font.family: theme.fontFamily @@ -329,11 +327,11 @@ ClippingRectangle { width: parent.width visible: text.length > 0 color: theme.subtext0 - elide: Text.ElideRight font.family: theme.fontFamily font.pixelSize: theme.tooltipFontPixelSize - 1 - text: root.modelDetails(modelRow.modelData) + text: root.modelSpec(modelRow.modelData) } + } } } diff --git a/main/scripts/llama_models.py b/main/scripts/llama_models.py index 79b1fb3..5a50421 100755 --- a/main/scripts/llama_models.py +++ b/main/scripts/llama_models.py @@ -1,6 +1,8 @@ #!/usr/bin/env python3 import json +import os +import subprocess import urllib.error import urllib.parse import urllib.request @@ -60,6 +62,56 @@ def decoded_tokens(slot): return int(next_token[0].get("n_decoded", 0)) +def model_pids(ports): + pids = {} + for entry in os.scandir("/proc"): + if not entry.name.isdigit(): + continue + try: + with open(f"{entry.path}/cmdline", "rb") as cmdline_file: + args = [ + part.decode(errors="replace") + for part in cmdline_file.read().split(b"\0") + if part + ] + port = model_port(args) + if port in ports: + pids[port] = int(entry.name) + except OSError: + continue + return pids + + +def gpu_memory_by_pid(pids): + if not pids: + return {} + try: + result = subprocess.run( + [ + "/usr/bin/nvidia-smi", + "--query-compute-apps=pid,used_gpu_memory", + "--format=csv,noheader,nounits", + ], + capture_output=True, + check=True, + text=True, + timeout=1, + ) + except (OSError, subprocess.SubprocessError): + return {} + + memory = {} + for line in result.stdout.splitlines(): + try: + pid_text, mib_text = line.split(",", 1) + pid = int(pid_text.strip()) + if pid in pids: + memory[pid] = memory.get(pid, 0) + int(mib_text.strip()) * 1048576 + except ValueError: + continue + return memory + + def main(): try: catalog = json.loads(fetch("/v1/models")).get("data", []) @@ -67,16 +119,22 @@ def main(): print(json.dumps({"models": [], "error": str(error)})) return - active_ports = connected_ports() - models = [] + running_items = [] for item in catalog: status_info = item.get("status", {}) status = str(status_info.get("value", "unloaded")) if status not in ("loaded", "loading"): continue + running_items.append( + (item, status_info, status, model_port(status_info.get("args", []))) + ) + pids = model_pids({port for _, _, _, port in running_items if port > 0}) + gpu_memory = gpu_memory_by_pid(set(pids.values())) + active_ports = connected_ports() + models = [] + for item, status_info, status, port in running_items: meta = item.get("meta", {}) - port = model_port(status_info.get("args", [])) has_active_request = status == "loaded" and port in active_ports model = { "id": str(item.get("id", "Unknown model")), @@ -87,8 +145,7 @@ def main(): "promptTokens": 0, "contextSize": int(meta.get("n_ctx", 0)), "params": float(meta.get("n_params", 0)), - "size": float(meta.get("size", 0)), - "modalities": item.get("architecture", {}).get("input_modalities", []), + "vramBytes": int(gpu_memory.get(pids.get(port), 0)), } if has_active_request: -- 2.51.2