Author docker-compose stack: llama.cpp (ROCm) + Open WebUI + Qdrant + Lazytainer
Resolves wayfinder ticket #4. Wires up the locked decisions from the map: - llama.cpp (ghcr.io/ggml-org/llama.cpp:server-rocm, gfx1201) serving Qwen3.8-27B-UD-Q4_K_XL.gguf, port published for direct Claude Code CLI / Kimi CLI access alongside Open WebUI. - Open WebUI with WEBUI_AUTH on, RAG+Memory wired to a standalone Qdrant service. - Lazytainer labels on llama-server for a 15 min idle-stop. - Named Docker volumes only (models, qdrant-data, openwebui-data) — no host bind-mounts. - One-off 'downloader' compose profile instead of a host-side script with its own dependencies, wrapped by scripts/download-model.sh. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,93 @@
|
||||
services:
|
||||
llama-server:
|
||||
image: ghcr.io/ggml-org/llama.cpp:server-rocm
|
||||
container_name: llama-server
|
||||
devices:
|
||||
- /dev/kfd
|
||||
- /dev/dri
|
||||
group_add:
|
||||
- video
|
||||
- render
|
||||
security_opt:
|
||||
- seccomp=unconfined
|
||||
ipc: host
|
||||
volumes:
|
||||
- models:/models
|
||||
command: >
|
||||
-m /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
|
||||
--host 0.0.0.0
|
||||
--port 8080
|
||||
--n-gpu-layers ${LLAMA_GPU_LAYERS:-999}
|
||||
--ctx-size ${LLAMA_CTX_SIZE:-65536}
|
||||
--jinja
|
||||
ports:
|
||||
# published to the host so Claude Code CLI / Kimi CLI can reach it directly,
|
||||
# bypassing Open WebUI.
|
||||
- "${LLAMA_PORT:-8080}:8080"
|
||||
restart: unless-stopped
|
||||
networks: [ai-stack]
|
||||
labels:
|
||||
# ponytail: idle-timeout tuning lives here, not in a separate lazytainer config file —
|
||||
# one place to look. Raise LAZYTAINER_INACTIVE_TIMEOUT if 15 min proves too eager.
|
||||
- "lazytainer.group.llamaserver.sleepMethod=stop"
|
||||
- "lazytainer.group.llamaserver.ports=8080"
|
||||
- "lazytainer.group.llamaserver.inactiveTimeout=${LAZYTAINER_INACTIVE_TIMEOUT:-900}"
|
||||
- "lazytainer.group.llamaserver.minPacketThreshold=2"
|
||||
|
||||
# ponytail: one-off downloader, not a standing service — run via
|
||||
# `docker compose --profile tools run --rm downloader` (see scripts/download-model.sh).
|
||||
# Keeps the model file inside the named `models` volume instead of a host bind-mount.
|
||||
downloader:
|
||||
image: curlimages/curl:latest
|
||||
profiles: ["tools"]
|
||||
volumes:
|
||||
- models:/models
|
||||
entrypoint: ["sh", "-c"]
|
||||
command:
|
||||
- >
|
||||
curl -L --fail --create-dirs -o /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
|
||||
https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
|
||||
|
||||
qdrant:
|
||||
image: qdrant/qdrant:latest
|
||||
container_name: qdrant
|
||||
volumes:
|
||||
- qdrant-data:/qdrant/storage
|
||||
restart: unless-stopped
|
||||
networks: [ai-stack]
|
||||
|
||||
open-webui:
|
||||
image: ghcr.io/open-webui/open-webui:main
|
||||
container_name: open-webui
|
||||
depends_on:
|
||||
- qdrant
|
||||
volumes:
|
||||
- openwebui-data:/app/backend/data
|
||||
environment:
|
||||
- WEBUI_AUTH=True
|
||||
- OPENAI_API_BASE_URL=http://llama-server:8080/v1
|
||||
- OPENAI_API_KEY=${OPENAI_API_KEY:-local}
|
||||
- VECTOR_DB=qdrant
|
||||
- QDRANT_URI=http://qdrant:6333
|
||||
ports:
|
||||
- "${WEBUI_PORT:-3000}:8080"
|
||||
restart: unless-stopped
|
||||
networks: [ai-stack]
|
||||
|
||||
lazytainer:
|
||||
image: ghcr.io/vmorganp/lazytainer:master
|
||||
container_name: lazytainer
|
||||
network_mode: host
|
||||
volumes:
|
||||
- /var/run/docker.sock:/var/run/docker.sock:ro
|
||||
restart: unless-stopped
|
||||
depends_on:
|
||||
- llama-server
|
||||
|
||||
networks:
|
||||
ai-stack:
|
||||
|
||||
volumes:
|
||||
models:
|
||||
qdrant-data:
|
||||
openwebui-data:
|
||||
Reference in New Issue
Block a user