From 4a99ac5164f8b50c961bf6dc3e8b625f8acf495b Mon Sep 17 00:00:00 2001 From: joelazar Date: Fri, 17 Jul 2026 12:32:58 +0200 Subject: [PATCH] :sparkles: migrate to llama.cpp for local inference - Add llama.cpp package to Brewfile.private and Brewfile.work; remove ollama - Add fish functions for llama-server lifecycle management (start, stop, restart, status, logs) - Add com.joelazar.llama-server LaunchAgent with M4 Pro tuning and Qwen3.6-35B-A3B model - Update pi agent provider from ollama to llama-cpp with new model alias - Update README documentation to reference llama.cpp instead of Ollama - Remove com.joelazar.ollama LaunchAgent configuration --- Brewfile.private | 2 +- Brewfile.work | 4 +- README.md | 2 +- .../functions/llama-cpp-logs.fish | 3 + .../functions/llama-cpp-restart.fish | 4 + .../functions/llama-cpp-start.fish | 4 + .../functions/llama-cpp-status.fish | 18 +++ .../functions/llama-cpp-stop.fish | 4 + dot_pi/agent/models.json | 6 +- .../com.joelazar.llama-server.plist | 114 ++++++++++++++++++ .../LaunchAgents/com.joelazar.ollama.plist | 91 -------------- 11 files changed, 154 insertions(+), 98 deletions(-) create mode 100644 dot_config/private_fish/functions/llama-cpp-logs.fish create mode 100644 dot_config/private_fish/functions/llama-cpp-restart.fish create mode 100644 dot_config/private_fish/functions/llama-cpp-start.fish create mode 100644 dot_config/private_fish/functions/llama-cpp-status.fish create mode 100644 dot_config/private_fish/functions/llama-cpp-stop.fish create mode 100644 private_Library/LaunchAgents/com.joelazar.llama-server.plist delete mode 100644 private_Library/LaunchAgents/com.joelazar.ollama.plist diff --git a/Brewfile.private b/Brewfile.private index e21c1342..8dfbe5bb 100644 --- a/Brewfile.private +++ b/Brewfile.private @@ -74,6 +74,7 @@ brew "kubernetes-cli" brew "kubectx" brew "lazydocker" brew "lazygit" +brew "llama.cpp" brew "llvm" brew "luajit" brew "luarocks" @@ -82,7 +83,6 @@ brew "mpv" brew "ncdu" brew "neovim" brew "nss" -brew "ollama" brew "ouch" brew "pngpaste" brew "poetry" diff --git a/Brewfile.work b/Brewfile.work index c9968e16..92a2d42d 100644 --- a/Brewfile.work +++ b/Brewfile.work @@ -176,6 +176,8 @@ brew "qemu" brew "lima-additional-guestagents" # CLI for SQLite Databases with auto-completion and syntax highlighting brew "litecli" +# LLM inference in C/C++ +brew "llama.cpp" # Next-gen compiler infrastructure brew "llvm" # Just-In-Time Compiler (JIT) for the Lua programming language @@ -198,8 +200,6 @@ brew "tree-sitter" brew "neovim" # Libraries for security-enabled client and server applications brew "nss" -# Create, run, and share large language models (LLMs) -brew "ollama" # Drop-in replacement for Terraform. Infrastructure as Code Tool brew "opentofu" # Painless compression and decompression for your terminal diff --git a/README.md b/README.md index 8bf4297d..8f30fabf 100644 --- a/README.md +++ b/README.md @@ -144,7 +144,7 @@ This repo currently tracks config for: - a separate Claude work profile in [`dot_claude-work/`](dot_claude-work/) - [Pi](https://github.com/mariozechner/pi-coding-agent) in [`dot_pi/`](dot_pi/) - [Codex](https://github.com/openai/codex) templates in [`dot_codex/`](dot_codex/) -- [Ollama](https://ollama.com/) for local models +- [llama.cpp](https://github.com/ggml-org/llama.cpp) (`llama-server`) for local models There is also a small helper script, [`ai-update`](private_dot_local/bin/executable_ai-update), that updates the main CLI agents and Pi extensions. diff --git a/dot_config/private_fish/functions/llama-cpp-logs.fish b/dot_config/private_fish/functions/llama-cpp-logs.fish new file mode 100644 index 00000000..0640b595 --- /dev/null +++ b/dot_config/private_fish/functions/llama-cpp-logs.fish @@ -0,0 +1,3 @@ +function llama-cpp-logs --description "Tail the llama-server logs" + tail -f /tmp/com.joelazar.llama-server.out /tmp/com.joelazar.llama-server.err +end diff --git a/dot_config/private_fish/functions/llama-cpp-restart.fish b/dot_config/private_fish/functions/llama-cpp-restart.fish new file mode 100644 index 00000000..966d3769 --- /dev/null +++ b/dot_config/private_fish/functions/llama-cpp-restart.fish @@ -0,0 +1,4 @@ +function llama-cpp-restart --description "Restart the llama-server launchd agent" + launchctl kickstart -k gui/(id -u)/com.joelazar.llama-server + and echo "llama-server restarted" +end diff --git a/dot_config/private_fish/functions/llama-cpp-start.fish b/dot_config/private_fish/functions/llama-cpp-start.fish new file mode 100644 index 00000000..58776d34 --- /dev/null +++ b/dot_config/private_fish/functions/llama-cpp-start.fish @@ -0,0 +1,4 @@ +function llama-cpp-start --description "Start the llama-server launchd agent" + launchctl bootstrap gui/(id -u) ~/Library/LaunchAgents/com.joelazar.llama-server.plist + and echo "llama-server started (http://127.0.0.1:11434)" +end diff --git a/dot_config/private_fish/functions/llama-cpp-status.fish b/dot_config/private_fish/functions/llama-cpp-status.fish new file mode 100644 index 00000000..fa52279b --- /dev/null +++ b/dot_config/private_fish/functions/llama-cpp-status.fish @@ -0,0 +1,18 @@ +function llama-cpp-status --description "Show llama-server agent and API status" + set -l state (launchctl print gui/(id -u)/com.joelazar.llama-server 2>/dev/null | string match -r 'state = \S+' | head -1) + if test -n "$state" + echo "agent: $state" + else + echo "agent: not loaded" + return 1 + end + + set -l health (curl -s --max-time 2 http://127.0.0.1:11434/health) + if string match -q '*"ok"*' $health + echo "api: up — "(curl -s --max-time 2 http://127.0.0.1:11434/v1/models | jq -r '.data[].id' | string join ', ') + else if test -n "$health" + echo "api: $health" + else + echo "api: not responding" + end +end diff --git a/dot_config/private_fish/functions/llama-cpp-stop.fish b/dot_config/private_fish/functions/llama-cpp-stop.fish new file mode 100644 index 00000000..32615072 --- /dev/null +++ b/dot_config/private_fish/functions/llama-cpp-stop.fish @@ -0,0 +1,4 @@ +function llama-cpp-stop --description "Stop the llama-server launchd agent (until next login or llama-cpp-start)" + launchctl bootout gui/(id -u)/com.joelazar.llama-server + and echo "llama-server stopped" +end diff --git a/dot_pi/agent/models.json b/dot_pi/agent/models.json index 3904c200..7a2ea20c 100644 --- a/dot_pi/agent/models.json +++ b/dot_pi/agent/models.json @@ -1,9 +1,9 @@ { "providers": { - "ollama": { + "llama-cpp": { "baseUrl": "http://localhost:11434/v1", "api": "openai-completions", - "apiKey": "ollama", + "apiKey": "llama-cpp", "compat": { "supportsDeveloperRole": false, "supportsReasoningEffort": false, @@ -11,7 +11,7 @@ }, "models": [ { - "id": "qwen3.6:latest", + "id": "qwen3.6-35b-a3b", "name": "Qwen 3.6 35B-A3B (Local)", "reasoning": true, "input": ["text", "image"], diff --git a/private_Library/LaunchAgents/com.joelazar.llama-server.plist b/private_Library/LaunchAgents/com.joelazar.llama-server.plist new file mode 100644 index 00000000..92afcaff --- /dev/null +++ b/private_Library/LaunchAgents/com.joelazar.llama-server.plist @@ -0,0 +1,114 @@ + + + + + + Label + com.joelazar.llama-server + + ProgramArguments + + /opt/homebrew/bin/llama-server + + + -hf + unsloth/Qwen3.6-35B-A3B-MTP-GGUF:UD-Q5_K_XL + + + --alias + qwen3.6-35b-a3b + + + --host + 127.0.0.1 + --port + 11434 + + + -c + 65536 + + + -fa + on + + + -ctk + q8_0 + -ctv + q8_0 + + + -ngl + 99 + + + -np + 1 + + + --jinja + + + EnvironmentVariables + + + PATH + /opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin + + + HOME + /Users/joelazar + + + + RunAtLoad + + + KeepAlive + + + + ThrottleInterval + 10 + + StandardOutPath + /tmp/com.joelazar.llama-server.out + + StandardErrorPath + /tmp/com.joelazar.llama-server.err + + ProcessType + Interactive + + diff --git a/private_Library/LaunchAgents/com.joelazar.ollama.plist b/private_Library/LaunchAgents/com.joelazar.ollama.plist deleted file mode 100644 index 1b6e635b..00000000 --- a/private_Library/LaunchAgents/com.joelazar.ollama.plist +++ /dev/null @@ -1,91 +0,0 @@ - - - - - - Label - com.joelazar.ollama - - ProgramArguments - - /opt/homebrew/bin/ollama - serve - - - EnvironmentVariables - - - OLLAMA_CONTEXT_LENGTH - 65536 - - - OLLAMA_FLASH_ATTENTION - 1 - - - OLLAMA_KV_CACHE_TYPE - q8_0 - - - OLLAMA_KEEP_ALIVE - 30m - - - OLLAMA_NUM_PARALLEL - 1 - - OLLAMA_MAX_LOADED_MODELS - 1 - - - PATH - /opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin - - - HOME - /Users/joelazar - - - - RunAtLoad - - - KeepAlive - - - - ThrottleInterval - 10 - - StandardOutPath - /tmp/com.joelazar.ollama.out - - StandardErrorPath - /tmp/com.joelazar.ollama.err - - ProcessType - Interactive - - -- 2.51.2