services: llama-server: image: ghcr.io/ggml-org/llama.cpp:server-rocm container_name: llama-server devices: - /dev/kfd - /dev/dri group_add: - video - render security_opt: - seccomp=unconfined ipc: host volumes: - models:/models command: > -m /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf} --host 0.0.0.0 --port 8080 --n-gpu-layers ${LLAMA_GPU_LAYERS:-999} --ctx-size ${LLAMA_CTX_SIZE:-65536} --jinja ports: # published to the host so Claude Code CLI / Kimi CLI can reach it directly, # bypassing Open WebUI. - "${LLAMA_PORT:-8080}:8080" restart: unless-stopped networks: [ai-stack] labels: # ponytail: idle-timeout tuning lives here, not in a separate lazytainer config file — # one place to look. Raise LAZYTAINER_INACTIVE_TIMEOUT if 15 min proves too eager. - "lazytainer.group.llamaserver.sleepMethod=stop" - "lazytainer.group.llamaserver.ports=8080" - "lazytainer.group.llamaserver.inactiveTimeout=${LAZYTAINER_INACTIVE_TIMEOUT:-900}" - "lazytainer.group.llamaserver.minPacketThreshold=2" # ponytail: one-off downloader, not a standing service — run via # `docker compose --profile tools run --rm downloader` (see scripts/download-model.sh). # Keeps the model file inside the named `models` volume instead of a host bind-mount. downloader: image: curlimages/curl:latest profiles: ["tools"] volumes: - models:/models entrypoint: ["sh", "-c"] command: - > curl -L --fail --create-dirs -o /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf} https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf} qdrant: image: qdrant/qdrant:latest container_name: qdrant volumes: - qdrant-data:/qdrant/storage restart: unless-stopped networks: [ai-stack] open-webui: image: ghcr.io/open-webui/open-webui:main container_name: open-webui depends_on: - qdrant volumes: - openwebui-data:/app/backend/data environment: - WEBUI_AUTH=True - OPENAI_API_BASE_URL=http://llama-server:8080/v1 - OPENAI_API_KEY=${OPENAI_API_KEY:-local} - VECTOR_DB=qdrant - QDRANT_URI=http://qdrant:6333 ports: - "${WEBUI_PORT:-3000}:8080" restart: unless-stopped networks: [ai-stack] lazytainer: image: ghcr.io/vmorganp/lazytainer:master container_name: lazytainer network_mode: host volumes: - /var/run/docker.sock:/var/run/docker.sock:ro restart: unless-stopped depends_on: - llama-server networks: ai-stack: volumes: models: qdrant-data: openwebui-data: