services: llama-server: image: ghcr.io/ggml-org/llama.cpp:server-rocm container_name: llama-server devices: - /dev/kfd - /dev/dri group_add: - video - render security_opt: - seccomp=unconfined ipc: host volumes: - models:/models command: > -m /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf} --host 0.0.0.0 --port 8080 --n-gpu-layers ${LLAMA_GPU_LAYERS:-999} --ctx-size ${LLAMA_CTX_SIZE:-65536} --jinja # No published host port: llama-server is reached only via the litellm # proxy on the ai-stack docker network now — see issue #15. Its # unauthenticated API no longer needs to be LAN-reachable directly. expose: - "8080" restart: unless-stopped networks: [ai-stack] labels: # ponytail: idle-timeout tuning lives here, not in a separate lazytainer config file — # one place to look. Raise LAZYTAINER_INACTIVE_TIMEOUT if 15 min proves too eager. - "lazytainer.group.llamaserver.sleepMethod=stop" - "lazytainer.group.llamaserver.ports=8080" - "lazytainer.group.llamaserver.inactiveTimeout=${LAZYTAINER_INACTIVE_TIMEOUT:-900}" - "lazytainer.group.llamaserver.minPacketThreshold=2" # ponytail: one-off downloader, not a standing service — run via # `docker compose --profile tools run --rm downloader` (see scripts/download-model.sh). # Keeps the model file inside the named `models` volume instead of a host bind-mount. downloader: image: curlimages/curl:latest profiles: ["tools"] volumes: - models:/models entrypoint: ["sh", "-c"] command: - > curl -L --fail --create-dirs -o /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf} https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf} qdrant: image: qdrant/qdrant:latest container_name: qdrant volumes: - qdrant-data:/qdrant/storage restart: unless-stopped networks: [ai-stack] healthcheck: test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/6333'"] interval: 10s timeout: 5s retries: 5 open-webui: image: ghcr.io/open-webui/open-webui:main container_name: open-webui depends_on: qdrant: condition: service_healthy litellm: condition: service_healthy volumes: - openwebui-data:/app/backend/data env_file: .env environment: - WEBUI_AUTH=True # Routed through the litellm proxy, not llama-server directly — see issue #15. # OPENAI_API_KEY must be a virtual key created for Open WebUI per # docs/proxy-key-onboarding.md (name it "openwebui"), set as # OPENWEBUI_LITELLM_KEY in .env. - OPENAI_API_BASE_URL=http://litellm:4000/v1 - OPENAI_API_KEY=${OPENWEBUI_LITELLM_KEY} - VECTOR_DB=qdrant - QDRANT_URI=http://qdrant:6333 ports: - "${WEBUI_PORT:-8008}:8080" restart: unless-stopped networks: [ai-stack] litellm: image: ghcr.io/berriai/litellm:main-stable container_name: litellm depends_on: litellm-db: condition: service_healthy llama-server: condition: service_started volumes: - ./litellm-config.yaml:/app/config.yaml:ro # LITELLM_MASTER_KEY / LITELLM_SALT_KEY come straight from .env via env_file # (names match what litellm reads). LITELLM_SALT_KEY must not change after # first run — see .env.example. env_file: .env environment: - DATABASE_URL=postgresql://litellm:${LITELLM_DB_PASSWORD}@litellm-db:5432/litellm command: ["--config", "/app/config.yaml", "--port", "4000"] ports: # published for LAN access (proxy.ai.home) and, via NPM, proxy.ai.haylan.ch — # NPM must deny the /ui path on the external host. See docs/network-access.md. - "${LITELLM_PORT:-4000}:4000" restart: unless-stopped networks: [ai-stack] healthcheck: test: - CMD-SHELL - python3 -c "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')" interval: 30s timeout: 10s retries: 3 start_period: 40s litellm-db: image: postgres:16-alpine container_name: litellm-db env_file: .env environment: - POSTGRES_USER=litellm - POSTGRES_PASSWORD=${LITELLM_DB_PASSWORD} - POSTGRES_DB=litellm volumes: - litellm-db-data:/var/lib/postgresql/data restart: unless-stopped networks: [ai-stack] healthcheck: test: ["CMD-SHELL", "pg_isready -d litellm -U litellm"] interval: 5s timeout: 5s retries: 10 lazytainer: image: ghcr.io/vmorganp/lazytainer:master container_name: lazytainer network_mode: host volumes: - /var/run/docker.sock:/var/run/docker.sock:ro restart: unless-stopped depends_on: - llama-server networks: ai-stack: volumes: models: qdrant-data: openwebui-data: litellm-db-data: