diff --git a/main/modules/LlamaModels.qml b/main/modules/LlamaModels.qml index d07295b..7dd99bd 100644 --- a/main/modules/LlamaModels.qml +++ b/main/modules/LlamaModels.qml @@ -67,7 +67,7 @@ ClippingRectangle { let state = generationSamples[model.id]; if (!state || state.generationId !== model.generationId) - state = { generationId: model.generationId, samples: [], phase: "" }; + state = { generationId: model.generationId, samples: [], phase: "prompt" }; const sample = { timestamp: now, @@ -89,12 +89,10 @@ ClippingRectangle { const generatedTokens = Math.max(0, sample.generated - first.generated); if (generatedTokens > 0) state.phase = "generation"; - else if (promptTokens > 0 || (state.phase.length === 0 && sample.prompt > 0)) - state.phase = "prompt"; const tokens = state.phase === "prompt" ? promptTokens : generatedTokens; model.tokensPerSecond = elapsed > 0 ? Math.max(0, tokens) * 1000 / elapsed : 0; - model.ratePhase = state.phase || "generation"; + model.ratePhase = state.phase; } function rateText(value) { @@ -131,26 +129,25 @@ ClippingRectangle { function paramsText(value) { const params = Number(value || 0); - return params > 0 ? `${(params / 1000000000).toFixed(1)}B params` : ""; + return params > 0 ? `${(params / 1000000000).toFixed(1)}B param` : ""; } - function sizeText(value) { - const size = Number(value || 0); - return size > 0 ? `${(size / 1073741824).toFixed(1)} GiB` : ""; + function contextText(value) { + const context = Number(value || 0); + return context > 0 ? `${Math.round(context / 1024)}k ctx` : ""; } - function modelDetails(model) { + function modelSpec(model) { const details = []; + const vram = Number(model.vramBytes || 0); const params = paramsText(model.params); - const size = sizeText(model.size); + const context = contextText(model.contextSize); + if (vram > 0) + details.push(`${(vram / 1073741824).toFixed(1)} GiB VRAM`); if (params.length > 0) details.push(params); - if (size.length > 0) - details.push(size); - if (Number(model.contextSize || 0) > 0) - details.push(`${Number(model.contextSize).toLocaleString()} ctx`); - if (Array.isArray(model.modalities) && model.modalities.length > 0) - details.push(model.modalities.join(" + ")); + if (context.length > 0) + details.push(context); return details.join(" ยท "); } @@ -304,9 +301,10 @@ ClippingRectangle { Row { width: parent.width + spacing: 8 Text { - width: parent.width - stateLabel.width + width: parent.width - stateLabel.width - parent.spacing color: theme.text elide: Text.ElideMiddle font.family: theme.fontFamily @@ -329,11 +327,11 @@ ClippingRectangle { width: parent.width visible: text.length > 0 color: theme.subtext0 - elide: Text.ElideRight font.family: theme.fontFamily font.pixelSize: theme.tooltipFontPixelSize - 1 - text: root.modelDetails(modelRow.modelData) + text: root.modelSpec(modelRow.modelData) } + } } } diff --git a/main/scripts/llama_models.py b/main/scripts/llama_models.py index 79b1fb3..5a50421 100755 --- a/main/scripts/llama_models.py +++ b/main/scripts/llama_models.py @@ -1,6 +1,8 @@ #!/usr/bin/env python3 import json +import os +import subprocess import urllib.error import urllib.parse import urllib.request @@ -60,6 +62,56 @@ def decoded_tokens(slot): return int(next_token[0].get("n_decoded", 0)) +def model_pids(ports): + pids = {} + for entry in os.scandir("/proc"): + if not entry.name.isdigit(): + continue + try: + with open(f"{entry.path}/cmdline", "rb") as cmdline_file: + args = [ + part.decode(errors="replace") + for part in cmdline_file.read().split(b"\0") + if part + ] + port = model_port(args) + if port in ports: + pids[port] = int(entry.name) + except OSError: + continue + return pids + + +def gpu_memory_by_pid(pids): + if not pids: + return {} + try: + result = subprocess.run( + [ + "/usr/bin/nvidia-smi", + "--query-compute-apps=pid,used_gpu_memory", + "--format=csv,noheader,nounits", + ], + capture_output=True, + check=True, + text=True, + timeout=1, + ) + except (OSError, subprocess.SubprocessError): + return {} + + memory = {} + for line in result.stdout.splitlines(): + try: + pid_text, mib_text = line.split(",", 1) + pid = int(pid_text.strip()) + if pid in pids: + memory[pid] = memory.get(pid, 0) + int(mib_text.strip()) * 1048576 + except ValueError: + continue + return memory + + def main(): try: catalog = json.loads(fetch("/v1/models")).get("data", []) @@ -67,16 +119,22 @@ def main(): print(json.dumps({"models": [], "error": str(error)})) return - active_ports = connected_ports() - models = [] + running_items = [] for item in catalog: status_info = item.get("status", {}) status = str(status_info.get("value", "unloaded")) if status not in ("loaded", "loading"): continue + running_items.append( + (item, status_info, status, model_port(status_info.get("args", []))) + ) + pids = model_pids({port for _, _, _, port in running_items if port > 0}) + gpu_memory = gpu_memory_by_pid(set(pids.values())) + active_ports = connected_ports() + models = [] + for item, status_info, status, port in running_items: meta = item.get("meta", {}) - port = model_port(status_info.get("args", [])) has_active_request = status == "loaded" and port in active_ports model = { "id": str(item.get("id", "Unknown model")), @@ -87,8 +145,7 @@ def main(): "promptTokens": 0, "contextSize": int(meta.get("n_ctx", 0)), "params": float(meta.get("n_params", 0)), - "size": float(meta.get("size", 0)), - "modalities": item.get("architecture", {}).get("input_modalities", []), + "vramBytes": int(gpu_memory.get(pids.get(port), 0)), } if has_active_request: