Files
LLM-Server/docker-compose.yml
T
haylanandClaude-Bot 20cc0bcc70 fix(llm): point qwen-classifier at the Q8 GGUF already on disk
The Q4_K_M-class file this originally specced didn't exist yet on
gameserver (classifier crash-looped: "No such file or directory").
A Q8_K_XL GGUF for the same model was already sitting in the models
volume from something earlier — point at that instead of downloading a
new file, and drop the KV cache quant to q4_0/q4_0 to keep total RAM
comfortable now that the weights are the larger Q8 variant.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-09 16:37:54 +02:00

334 lines
15 KiB
YAML

services:
llama-server:
image: ghcr.io/ggml-org/llama.cpp:server-rocm
container_name: llama-server
devices:
- /dev/kfd
- /dev/dri
# Numeric GIDs, not names — see HOST_VIDEO_GID/HOST_RENDER_GID in
# .env.example and docs/research/rocm-gpu-pin-and-render-group.md.
group_add:
- "${HOST_VIDEO_GID:?run scripts/update.sh first to resolve this}"
- "${HOST_RENDER_GID:?run scripts/update.sh first to resolve this}"
security_opt:
- seccomp=unconfined
ipc: host
# Caps this process's HIP hardware-queue allocation — works around
# ROCm/ROCm#5706 (GPU pinned at 100%/boost-clock whenever two
# concurrent HIP contexts touch this card). See the research doc above.
environment:
- GPU_MAX_HW_QUEUES=1
volumes:
- models:/models
command: >
-m /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
--host 0.0.0.0
--port 8080
--n-gpu-layers ${LLAMA_GPU_LAYERS:-999}
--ctx-size ${LLAMA_CTX_SIZE:-262144}
--parallel ${LLAMA_PARALLEL:-2}
--flash-attn on
--cache-type-k q8_0
--cache-type-v q8_0
--jinja
# No published host port: llama-server is reached only via the omniroute
# gateway on the ai-stack docker network now — see issue #15. Its
# unauthenticated API no longer needs to be LAN-reachable directly.
expose:
- "8080"
restart: unless-stopped
networks: [ai-stack]
labels:
# ponytail: idle-timeout tuning lives here, not in a separate lazytainer config file —
# one place to look. Raise LAZYTAINER_INACTIVE_TIMEOUT if 15 min proves too eager.
- "lazytainer.group.llamaserver.sleepMethod=stop"
- "lazytainer.group.llamaserver.ports=8080"
- "lazytainer.group.llamaserver.inactiveTimeout=${LAZYTAINER_INACTIVE_TIMEOUT:-900}"
- "lazytainer.group.llamaserver.minPacketThreshold=2"
# Dedicated backend for qwen-code's tool-call harmfulness classifier
# (fastModel in ~/.qwen/settings.json). Was aliased onto llama-server's own
# 27B connection — every classification call then queued behind whatever
# heavy generation was already running on that model's 2 GPU slots (issue
# tracker: OmniRoute semaphore/pr-agent investigation). CPU-only, own
# process, own queue: structurally can't contend with llama-server for a
# GPU slot. Needs >=131072 ctx (qwen-code requirement); Qwen3-4B-Instruct-2507
# is the smallest Qwen3 that supports that natively (262144) without
# RoPE-scaling — the smaller 0.6B/1.7B/4B (non-2507) models only go to
# 40960. Reuses a Q8_K_XL GGUF already sitting in the models volume from
# something earlier (better weight quality than the Q4_K_M-class file
# originally specced here, no download needed). KV cache dropped to
# q4_0/q4_0 to compensate: full-context q8_0/q8_0 (~9.6GiB) on top of the
# larger Q8 weights (~4.7GiB) left too little slack against the rest of
# the stack (omniroute's 10g mem_limit, qdrant, neo4j) on gameserver's
# 31GiB total RAM; q4_0/q4_0 (~5.1GiB) + weights (~4.7GiB) ≈ 9.8GiB leaves
# comfortable headroom instead. Weight precision matters more than KV
# precision for a classification task, so this trade favors the weights.
qwen-classifier:
image: ghcr.io/ggml-org/llama.cpp:server
container_name: qwen-classifier
volumes:
- models:/models
command: >
-m /models/${LLAMA_CLASSIFIER_MODEL_FILE:-Qwen3-4B-Instruct-2507-UD-Q8_K_XL.gguf}
--host 0.0.0.0
--port 8080
--n-gpu-layers 0
--ctx-size 131072
--parallel 1
--cache-type-k q4_0
--cache-type-v q4_0
--jinja
expose:
- "8080"
restart: unless-stopped
networks: [ai-stack]
# ponytail: one-off downloader, not a standing service — run via
# `docker compose --profile tools run --rm downloader`. Folded into
# scripts/update.sh, which runs this every time; the `test -f` guard is
# what makes that safe to re-run without re-downloading. Keeps the model
# files inside the named `models` volume instead of a host bind-mount.
downloader:
image: curlimages/curl:latest
profiles: ["tools"]
# ponytail: named volume is created root-owned; curl_user (uid 100) can't
# write into it otherwise, so run as root for this one-off job.
user: root
volumes:
- models:/models
entrypoint: ["sh", "-c"]
command:
- >
test -f /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf} &&
echo "already downloaded, skipping" ||
curl -L --fail --create-dirs -o /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
# Same pattern as downloader above, separate service so this one small
# file doesn't get re-checked/re-pulled by the big model's job.
downloader-classifier:
image: curlimages/curl:latest
profiles: ["tools"]
user: root
volumes:
- models:/models
entrypoint: ["sh", "-c"]
command:
- >
test -f /models/${LLAMA_CLASSIFIER_MODEL_FILE:-Qwen3-4B-Instruct-2507-UD-Q4_K_XL.gguf} &&
echo "already downloaded, skipping" ||
curl -L --fail --create-dirs -o /models/${LLAMA_CLASSIFIER_MODEL_FILE:-Qwen3-4B-Instruct-2507-UD-Q4_K_XL.gguf}
https://huggingface.co/unsloth/Qwen3-4B-Instruct-2507-GGUF/resolve/main/${LLAMA_CLASSIFIER_MODEL_FILE:-Qwen3-4B-Instruct-2507-UD-Q4_K_XL.gguf}
# Fetches the three Qwen-Image FP8 files ComfyUI needs (diffusion model,
# text encoder, VAE) — same test -f guard pattern as downloader above.
# See docs/research/image-generation-model-choice.md and issue #42.
#
# ponytail: target paths assume ComfyUI's standard models/ layout under
# BASE_STORAGE_PATH (/storage) — same "not independently confirmed against
# the image's Dockerfile" caveat already flagged on the comfyui service
# below. If ComfyUI doesn't pick these up, check its actual models root
# first.
downloader-comfyui:
image: curlimages/curl:latest
profiles: ["tools"]
user: root
volumes:
- comfyui-data:/storage
entrypoint: ["sh", "-c"]
command:
- >
mkdir -p /storage/models/diffusion_models /storage/models/text_encoders /storage/models/vae &&
(test -f /storage/models/diffusion_models/${COMFYUI_DIFFUSION_MODEL_FILE:-qwen_image_fp8_e4m3fn.safetensors} &&
echo "diffusion model already downloaded, skipping" ||
curl -L --fail --create-dirs -o /storage/models/diffusion_models/${COMFYUI_DIFFUSION_MODEL_FILE:-qwen_image_fp8_e4m3fn.safetensors}
https://huggingface.co/Comfy-Org/Qwen-Image_ComfyUI/resolve/main/split_files/diffusion_models/${COMFYUI_DIFFUSION_MODEL_FILE:-qwen_image_fp8_e4m3fn.safetensors}) &&
(test -f /storage/models/text_encoders/${COMFYUI_TEXT_ENCODER_FILE:-qwen_2.5_vl_7b_fp8_scaled.safetensors} &&
echo "text encoder already downloaded, skipping" ||
curl -L --fail --create-dirs -o /storage/models/text_encoders/${COMFYUI_TEXT_ENCODER_FILE:-qwen_2.5_vl_7b_fp8_scaled.safetensors}
https://huggingface.co/Comfy-Org/Qwen-Image_ComfyUI/resolve/main/split_files/text_encoders/${COMFYUI_TEXT_ENCODER_FILE:-qwen_2.5_vl_7b_fp8_scaled.safetensors}) &&
(test -f /storage/models/vae/${COMFYUI_VAE_FILE:-qwen_image_vae.safetensors} &&
echo "vae already downloaded, skipping" ||
curl -L --fail --create-dirs -o /storage/models/vae/${COMFYUI_VAE_FILE:-qwen_image_vae.safetensors}
https://huggingface.co/Comfy-Org/Qwen-Image_ComfyUI/resolve/main/split_files/vae/${COMFYUI_VAE_FILE:-qwen_image_vae.safetensors})
# Local image generation — see issue #38 (wayfinder map). yurisasc's image
# is gfx1201-tuned specifically (R9700's arch), unlike the official/AMD
# ComfyUI image which doesn't pin RDNA4 support — see
# docs/research/image-generation-options.md.
comfyui:
image: yurisasc/comfyui-rocm7.1:latest
container_name: comfyui
devices:
- /dev/kfd
- /dev/dri
# Numeric GIDs, not names — see HOST_VIDEO_GID/HOST_RENDER_GID in
# .env.example and docs/research/rocm-gpu-pin-and-render-group.md.
group_add:
- "${HOST_VIDEO_GID:?run scripts/update.sh first to resolve this}"
- "${HOST_RENDER_GID:?run scripts/update.sh first to resolve this}"
security_opt:
- seccomp=unconfined
ipc: host
environment:
- HSA_OVERRIDE_GFX_VERSION=12.0.1
- PYTORCH_ROCM_ARCH=gfx1201
# This image also wants GID env vars directly (its own README asks
# for both these and group_add above) — same HOST_VIDEO_GID/
# HOST_RENDER_GID resolved by scripts/update.sh, shared with
# llama-server now instead of comfyui-only vars.
- PUID=${COMFYUI_PUID}
- PGID=${COMFYUI_PGID}
- VIDEO_GID=${HOST_VIDEO_GID}
- RENDER_GID=${HOST_RENDER_GID}
- BASE_STORAGE_PATH=/storage
volumes:
- comfyui-data:/storage
# ponytail: exact internal storage path taken from the image's own
# BASE_STORAGE_PATH env var, not independently confirmed against its
# Dockerfile — if models/workflows don't persist across a recreate,
# check this against the image's actual entrypoint first.
#
# Published host port (unlike llama-server's ai-stack-only pattern):
# ComfyUI's own UI is meant to be reachable directly too, for a planned
# external nginx reverse-proxy route to comfy.home — not just through
# OmniRoute. Still also reachable at http://comfyui:8188 internally on
# ai-stack, which is the URL to register as OmniRoute's comfyui
# provider (dashboard or POST /api/providers, per docs/proxy-key-onboarding.md
# — same undocumented-in-repo manual flow already used for llama-server).
ports:
- "8138:8188"
restart: unless-stopped
networks: [ai-stack]
# Replaces litellm — see issue #31 (wayfinder map) for the full migration
# rationale/findings. No static config.yaml equivalent: provider routing
# (llama-server, searxng-search) is registered once through the dashboard
# or POST /api/providers after first boot, not checked into this repo —
# see docs/proxy-key-onboarding.md.
omniroute:
image: diegosouzapw/omniroute:latest
container_name: omniroute
depends_on:
llama-server:
condition: service_started
volumes:
- omniroute-data:/app/data
env_file: .env
environment:
# Split-port mode: dashboard and API are fully separate ports (unlike
# LiteLLM's single :4000 for both /v1 and /ui) — both published
# directly below, unlike the old :4000-only host mapping.
- API_HOST=0.0.0.0
- API_PORT=${OMNIROUTE_API_PORT:-20129}
- DASHBOARD_PORT=${OMNIROUTE_DASHBOARD_PORT:-20128}
# Required to register llama-server/searxng-search as providers —
# their base URLs are LAN/container-internal addresses, blocked by
# default (SSRF guard against public-provider spoofing).
- OMNIROUTE_ALLOW_PRIVATE_PROVIDER_URLS=true
- OMNIROUTE_ALLOW_LOCAL_PROVIDER_URLS=true
# Required (production) per docs/reference/ENVIRONMENT.md — shared
# secret for the internal Codex Responses WebSocket bridge. Missed on
# first pass; docker-compose config validated fine without it, but
# the docs are explicit this one's required, not optional.
- OMNIROUTE_WS_BRIDGE_SECRET=${OMNIROUTE_WS_BRIDGE_SECRET}
# Default heap (1024MB) is dashboard-only sized per OmniRoute's own
# Docker guide — every client here is a coding CLI, which needs the
# larger figure the guide recommends. Paired with mem_limit below.
- OMNIROUTE_MEMORY_MB=8192
# Default 300000 (5 min) per OmniRoute's own docs, but this deployment
# had it dialed down elsewhere (dashboard) to ~95s — too tight for a
# contended local llama-server: large-context prefill under multiple
# concurrent slots can outrun that before the first SSE token arrives,
# so OmniRoute cancels a request that was actually still working (see
# LLAMA_PARALLEL above for the other half of this fix). Raised here so
# it's tracked in git instead of a dashboard-only setting.
- STREAM_IDLE_TIMEOUT_MS=${OMNIROUTE_STREAM_IDLE_TIMEOUT_MS:-180000}
# Same reasoning as litellm's extra_hosts entry below — ai-stack's bridge
# network can't resolve search.home on its own.
extra_hosts:
- "search.home:${SEARXNG_LAN_IP}"
ports:
- "${OMNIROUTE_API_PORT:-20129}:${OMNIROUTE_API_PORT:-20129}"
- "${OMNIROUTE_DASHBOARD_PORT:-20128}:${OMNIROUTE_DASHBOARD_PORT:-20128}"
# 10+ GiB ceiling per OmniRoute's Docker guide, matching
# OMNIROUTE_MEMORY_MB=8192 above.
mem_limit: 10g
# SQLite WAL needs time to checkpoint back into the main DB file on
# shutdown — the Docker guide's --stop-timeout 40 equivalent.
stop_grace_period: 40s
restart: unless-stopped
networks: [ai-stack]
# ponytail: TCP-connect check, not an HTTP /healthz GET — the image has
# no python3/curl/wget (confirmed live, `which` found only node), and
# OmniRoute's own Docker guide already treats a bare TCP probe on this
# port as an acceptable liveness check, not just the HTTP one. Simpler
# and avoids depending on /healthz's exact path/response shape.
healthcheck:
test:
- CMD-SHELL
- node -e "require('net').connect(${OMNIROUTE_API_PORT:-20129},'localhost').on('connect',function(){this.end();process.exit(0)}).on('error',()=>process.exit(1))"
interval: 30s
timeout: 10s
retries: 3
start_period: 40s
lazytainer:
image: ghcr.io/vmorganp/lazytainer:master
container_name: lazytainer
# NOT network_mode: host — lazytainer identifies its own container by
# matching os.Hostname() against the Docker container-ID list
# (vmorganp/Lazytainer, configureFromLabels()); under host networking the
# container inherits the host's hostname instead of its own ID, so that
# match always fails and it panics with "Could not determine container ID
# of lazytainer" on every start. Host networking also can't see traffic
# to llama-server:8080 anyway — that port only exists on the ai-stack
# bridge network (no host port published, see issue #15 above). Joining
# ai-stack instead fixes both: hostname becomes the real container ID,
# and it's on the same network as the traffic it's watching.
networks: [ai-stack]
volumes:
- /var/run/docker.sock:/var/run/docker.sock:ro
restart: unless-stopped
depends_on:
- llama-server
# RAG vector store — see docs/agents/... (wayfinder). Dashboard UI published
# directly like comfyui above, not gatewayed through omniroute (it isn't an
# LLM provider).
qdrant:
image: qdrant/qdrant:latest
container_name: qdrant
volumes:
- qdrant-data:/qdrant/storage
ports:
- "6333:6333"
restart: unless-stopped
networks: [ai-stack]
# RAG graph store, native vector index too (can absorb qdrant's job later
# if the two-DB split proves unnecessary — see wayfinder notes).
neo4j:
image: neo4j:5-community
container_name: neo4j
environment:
- NEO4J_AUTH=neo4j/${NEO4J_PASSWORD:?run scripts/update.sh first to resolve this}
volumes:
- neo4j-data:/data
ports:
- "7474:7474" # browser UI
- "7687:7687" # bolt
restart: unless-stopped
networks: [ai-stack]
networks:
ai-stack:
volumes:
models:
omniroute-data:
comfyui-data:
qdrant-data:
neo4j-data: