From 5cb34b19f36e7b51d322b2a4ffc4331a9e80f605 Mon Sep 17 00:00:00 2001 From: Haylan Date: Tue, 25 Aug 2026 07:18:03 +0200 Subject: [PATCH] Migrate Open WebUI and coding CLIs to the AI proxy (resolves #15) Open WebUI now points at litellm instead of llama-server directly, using a provisioned virtual key. llama-server's host port is dropped (internal-only on the ai-stack network) since the proxy is the only intended entry point now. docs/coding-cli-setup.md repointed at the proxy's endpoints/ports with per-CLI virtual keys instead of the old shared dummy key. Co-Authored-By: Claude Sonnet 5 --- .env.example | 6 +++--- README.md | 14 ++++++++---- docker-compose.yml | 17 +++++++++------ docs/coding-cli-setup.md | 46 +++++++++++++++++++++------------------- litellm-config.yaml | 5 ++++- 5 files changed, 52 insertions(+), 36 deletions(-) diff --git a/.env.example b/.env.example index e07498a..2ef2a24 100644 --- a/.env.example +++ b/.env.example @@ -12,9 +12,9 @@ LLAMA_PORT=8080 # --- Open WebUI --- WEBUI_PORT=3000 -# Dummy key — llama.cpp's OpenAI-compatible endpoint doesn't check it, but -# Open WebUI requires the field to be non-empty. -OPENAI_API_KEY=local +# Required — create an "openwebui" virtual key in LiteLLM's Admin UI first +# (see docs/proxy-key-onboarding.md), then paste it here. +OPENWEBUI_LITELLM_KEY= # --- Lazytainer --- # Seconds of inactivity before llama-server is stopped. 900 = 15 min. diff --git a/README.md b/README.md index 87d0077..4c20620 100644 --- a/README.md +++ b/README.md @@ -7,14 +7,20 @@ See the wayfinder map ([issue #1](https://git.arthurerlich.de/haylan/LLM-Server/ ## Quickstart ```bash -cp .env.example .env # adjust if needed +cp .env.example .env +# set LITELLM_MASTER_KEY / LITELLM_SALT_KEY (openssl rand -hex 32), see .env.example ./scripts/download-model.sh +docker compose up -d litellm litellm-db llama-server qdrant # bring the proxy up first +``` + +Log into LiteLLM's Admin UI (`http://:4000/ui`), create an `openwebui` virtual key (see [`docs/proxy-key-onboarding.md`](docs/proxy-key-onboarding.md)), set `OPENWEBUI_LITELLM_KEY` in `.env` to it, then: + +```bash docker compose up -d ``` - Open WebUI: `http://:3000` locally, or `ai.home` / `ai.haylan.ch` once routed through Nginx Proxy Manager — see [`docs/network-access.md`](docs/network-access.md). First signup becomes the admin account (`WEBUI_AUTH` is on). -- llama.cpp OpenAI-compatible API: `http://:8080/v1` — **LAN-only, not proxied**, see `docs/network-access.md`. -- llama.cpp Anthropic Messages API (for Claude Code CLI): `http://:8080/v1/messages` — same LAN-only scope. +- llama.cpp's own API is internal-only now — everything routes through the AI proxy below. Pointing Claude Code CLI, Kimi CLI, or OpenCode CLI at the local endpoint: see [`docs/coding-cli-setup.md`](docs/coding-cli-setup.md). @@ -29,4 +35,4 @@ An [AI gateway/proxy](https://git.arthurerlich.de/haylan/LLM-Server/issues/9) fr - Issuing a key for a new workload: [`docs/proxy-key-onboarding.md`](docs/proxy-key-onboarding.md). - Request priority across workloads: [`docs/proxy-request-priority.md`](docs/proxy-request-priority.md). -**Not yet done**: Open WebUI and the coding CLIs still talk to llama.cpp directly, not through this proxy — that migration is [issue #15](https://git.arthurerlich.de/haylan/LLM-Server/issues/15). **Not yet verified**: this config hasn't been smoke-tested on real hardware (LiteLLM's priority scheduler in particular is beta — see `docs/proxy-request-priority.md`) — see [issue #17](https://git.arthurerlich.de/haylan/LLM-Server/issues/17). +Open WebUI and the coding CLIs (see [`docs/coding-cli-setup.md`](docs/coding-cli-setup.md)) route through the proxy now — llama-server has no published host port anymore. **Not yet verified**: none of this has been smoke-tested on real hardware (LiteLLM's priority scheduler in particular is beta — see `docs/proxy-request-priority.md`) — see [issue #17](https://git.arthurerlich.de/haylan/LLM-Server/issues/17). diff --git a/docker-compose.yml b/docker-compose.yml index 48631e9..2828fde 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -20,10 +20,11 @@ services: --n-gpu-layers ${LLAMA_GPU_LAYERS:-999} --ctx-size ${LLAMA_CTX_SIZE:-65536} --jinja - ports: - # published to the host so Claude Code CLI / Kimi CLI can reach it directly, - # bypassing Open WebUI. - - "${LLAMA_PORT:-8080}:8080" + # No published host port: llama-server is reached only via the litellm + # proxy on the ai-stack docker network now — see issue #15. Its + # unauthenticated API no longer needs to be LAN-reachable directly. + expose: + - "8080" restart: unless-stopped networks: [ai-stack] labels: @@ -61,12 +62,16 @@ services: container_name: open-webui depends_on: - qdrant + - litellm volumes: - openwebui-data:/app/backend/data environment: - WEBUI_AUTH=True - - OPENAI_API_BASE_URL=http://llama-server:8080/v1 - - OPENAI_API_KEY=${OPENAI_API_KEY:-local} + # Routed through the litellm proxy, not llama-server directly — see issue #15. + # OPENAI_API_KEY must be a virtual key created for Open WebUI per + # docs/proxy-key-onboarding.md (name it "openwebui"), set in .env. + - OPENAI_API_BASE_URL=http://litellm:4000/v1 + - OPENAI_API_KEY=${OPENWEBUI_LITELLM_KEY:?set to the openwebui virtual key from LiteLLM's Admin UI} - VECTOR_DB=qdrant - QDRANT_URI=http://qdrant:6333 ports: diff --git a/docs/coding-cli-setup.md b/docs/coding-cli-setup.md index 101f033..8a74657 100644 --- a/docs/coding-cli-setup.md +++ b/docs/coding-cli-setup.md @@ -1,23 +1,25 @@ # Pointing a coding-agent CLI at this stack -This stack's llama.cpp server exposes two endpoints once `docker compose up` is running (see `docker-compose.yml`): +This stack routes through the [AI proxy](https://git.arthurerlich.de/haylan/LLM-Server/issues/9) (LiteLLM) rather than talking to llama.cpp directly — llama.cpp's own port is internal-only now (see `docker-compose.yml`). The proxy exposes: -- **OpenAI-compatible**: `http://:8080/v1` (or `${LLAMA_PORT}` if you changed it in `.env`) -- **Anthropic Messages API shim**: `http://:8080` (adds `/v1/messages`) +- **OpenAI-compatible**: `http://:4000/v1` (or `${LITELLM_PORT}` if you changed it in `.env`) +- **Anthropic Messages API** (LiteLLM's own unified `/v1/messages` endpoint, translating to the OpenAI-compatible backend): `http://:4000` -Both serve the same model — `Qwen3.8-27B-UD-Q4_K_XL.gguf` — behind whichever wire format the client speaks. +Both serve the same underlying model — `Qwen3.8-27B-UD-Q4_K_XL.gguf`, registered in the proxy as `qwen3.8-27b-local` — behind whichever wire format the client speaks. -`` is this machine's LAN address — its LAN IP, or `ai.home` if your local DNS resolves that hostname directly to the box. **This API is LAN-only, not reachable via `ai.haylan.ch`** — it's deliberately not registered in Nginx Proxy Manager (no auth of its own, unlike Open WebUI). See `docs/network-access.md`. If you're running a coding CLI from this machine itself, `localhost` works too. +`` is this machine's LAN address, or `proxy.ai.home` if your local DNS resolves that hostname directly to the box — see `docs/network-access.md`. If you're running a coding CLI from this machine itself, `localhost` works too. -> **Read this before relying on it for real work.** Qwen3.8-27B's tool-calling has **documented, open llama.cpp upstream bugs** (parser fails on text before ``, tool calls emitted as inert XML inside thinking blocks — see `docs/research/qwen3.8-27b-tool-calling.md`). Every setup below inherits this risk identically, regardless of which CLI or wire format you use. Don't trust it for unattended multi-step agentic work until you've run the smoke test in [issue #5](https://git.arthurerlich.de/haylan/LLM-Server/issues/5). +**Each CLI needs its own virtual key** — create one per docs/proxy-key-onboarding.md (LiteLLM's Admin UI, `-` naming, e.g. `claude-code-cli`, `kimi-cli`, `opencode-cli`). No budget set by default. These are the machine's interactive/high-priority workloads per `docs/proxy-request-priority.md`. + +> **Read this before relying on it for real work.** Qwen3.8-27B's tool-calling has **documented, open llama.cpp upstream bugs** (parser fails on text before ``, tool calls emitted as inert XML inside thinking blocks — see `docs/research/qwen3.8-27b-tool-calling.md`). Every setup below inherits this risk identically, regardless of which CLI or wire format you use. Don't trust it for unattended multi-step agentic work until you've run the smoke test in [issue #5](https://git.arthurerlich.de/haylan/LLM-Server/issues/5) (and the proxy-specific smoke test in [issue #17](https://git.arthurerlich.de/haylan/LLM-Server/issues/17)). ## Claude Code CLI -Claude Code speaks the **Anthropic Messages API** — point it at the shim, not the OpenAI-compatible endpoint: +Claude Code speaks the **Anthropic Messages API** — point it at the proxy's unified endpoint, not llama.cpp directly: ```bash -export ANTHROPIC_BASE_URL=http://:8080 -export ANTHROPIC_API_KEY=local # value is unchecked by llama.cpp, but the client requires it set +export ANTHROPIC_BASE_URL=http://:4000 +export ANTHROPIC_API_KEY= claude ``` @@ -25,13 +27,13 @@ Requires llama.cpp's `--jinja` flag (already set in `docker-compose.yml`) — wi ## Kimi CLI -Kimi CLI speaks plain **OpenAI Chat Completions** — no shim needed. Configure a provider block in its config file (`config.toml`): +Kimi CLI speaks plain **OpenAI Chat Completions**. Configure a provider block in its config file (`config.toml`): ```toml [providers.openai] type = "openai" -base_url = "http://:8080/v1" -api_key = "local" +base_url = "http://:4000/v1" +api_key = "" ``` If Kimi CLI's response parsing gets confused by Qwen's `...` reasoning tags, check its `reasoning_key` setting — it's configurable for non-standard local server responses. @@ -51,15 +53,15 @@ curl -fsSL https://opencode.ai/install | bash { "$schema": "https://opencode.ai/config.json", "provider": { - "llamacpp": { + "aiproxy": { "npm": "@ai-sdk/openai-compatible", - "name": "llama.cpp (local)", + "name": "AI proxy (local)", "options": { - "baseURL": "http://:8080/v1", - "apiKey": "sk-local-not-checked" + "baseURL": "http://:4000/v1", + "apiKey": "" }, "models": { - "qwen3.8-27b": { + "qwen3.8-27b-local": { "name": "Qwen3.8-27B", "limit": { "context": 65536, "output": 8192 } } @@ -71,7 +73,7 @@ curl -fsSL https://opencode.ai/install | bash Set `limit.context` to match whatever `LLAMA_CTX_SIZE` this stack is actually running with (`.env`), not a value assumed from the model card — OpenCode uses it for its own context-management bookkeeping, not the server. -Select the model with `llamacpp/qwen3.8-27b`. +Select the model with `aiproxy/qwen3.8-27b-local`. **OpenCode-specific risks** (on top of the shared Qwen3.8-27B tool-calling risk above): - Requires llama.cpp's `--jinja` flag (already set) — without it, OpenCode's unconditional tool-call scaffolding gets a 500. @@ -82,8 +84,8 @@ Select the model with `llamacpp/qwen3.8-27b`. | CLI | Wire format | Endpoint | Config | |---|---|---|---| -| Claude Code | Anthropic Messages | `http://:8080` | `ANTHROPIC_BASE_URL` env var | -| Kimi CLI | OpenAI Chat Completions | `http://:8080/v1` | `config.toml` provider block | -| OpenCode | OpenAI Chat Completions | `http://:8080/v1` | `opencode.json` provider block | +| Claude Code | Anthropic Messages | `http://:4000` | `ANTHROPIC_BASE_URL` env var | +| Kimi CLI | OpenAI Chat Completions | `http://:4000/v1` | `config.toml` provider block | +| OpenCode | OpenAI Chat Completions | `http://:4000/v1` | `opencode.json` provider block | -Further reading: `docs/research/qwen3.8-27b-tool-calling.md`, `docs/research/opencode-cli-setup.md`. +Further reading: `docs/research/qwen3.8-27b-tool-calling.md`, `docs/research/opencode-cli-setup.md`, `docs/proxy-key-onboarding.md`. diff --git a/litellm-config.yaml b/litellm-config.yaml index 93b5bb3..c3e415e 100644 --- a/litellm-config.yaml +++ b/litellm-config.yaml @@ -1,7 +1,10 @@ model_list: - model_name: qwen3.8-27b-local litellm_params: - model: openai/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf} + # Static name — llama.cpp serves whatever model it loaded regardless of + # what's requested here; this string isn't shell-expanded (this file + # isn't docker-compose.yml, .env vars don't reach it). + model: openai/qwen3.8-27b-local api_base: http://llama-server:8080/v1 api_key: local model_info: