lazytainer panics on every start with 'Could not determine container ID of lazytainer': its self-detection (vmorganp/Lazytainer, configureFromLabels()) matches os.Hostname() against the Docker container-ID list, but network_mode: host makes the container inherit the host machine's hostname instead of its own container ID, so the match never succeeds. Upstream's own example compose file doesn't use host networking (default bridge + published ports) — nothing about lazytainer requires it. It also couldn't have seen the traffic it's meant to watch: llama-server has no published host port (issue #15) and only exists on the ai-stack bridge network, which host networking has no visibility into. Joining ai-stack instead fixes both problems at once. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_018WHfjWrSEcGhCoeu6dQfDa
305 lines
11 KiB
YAML
305 lines
11 KiB
YAML
services:
|
|
llama-server:
|
|
image: ghcr.io/ggml-org/llama.cpp:server-rocm
|
|
container_name: llama-server
|
|
devices:
|
|
- /dev/kfd
|
|
- /dev/dri
|
|
group_add:
|
|
- video
|
|
- render
|
|
security_opt:
|
|
- seccomp=unconfined
|
|
ipc: host
|
|
volumes:
|
|
- models:/models
|
|
command: >
|
|
-m /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
|
|
--host 0.0.0.0
|
|
--port 8080
|
|
--n-gpu-layers ${LLAMA_GPU_LAYERS:-999}
|
|
--ctx-size ${LLAMA_CTX_SIZE:-65536}
|
|
--jinja
|
|
# No published host port: llama-server is reached only via the litellm
|
|
# proxy on the ai-stack docker network now — see issue #15. Its
|
|
# unauthenticated API no longer needs to be LAN-reachable directly.
|
|
expose:
|
|
- "8080"
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
labels:
|
|
# ponytail: idle-timeout tuning lives here, not in a separate lazytainer config file —
|
|
# one place to look. Raise LAZYTAINER_INACTIVE_TIMEOUT if 15 min proves too eager.
|
|
- "lazytainer.group.llamaserver.sleepMethod=stop"
|
|
- "lazytainer.group.llamaserver.ports=8080"
|
|
- "lazytainer.group.llamaserver.inactiveTimeout=${LAZYTAINER_INACTIVE_TIMEOUT:-900}"
|
|
- "lazytainer.group.llamaserver.minPacketThreshold=2"
|
|
|
|
embedding-server:
|
|
image: ghcr.io/ggml-org/llama.cpp:server-rocm
|
|
container_name: embedding-server
|
|
devices:
|
|
- /dev/kfd
|
|
- /dev/dri
|
|
group_add:
|
|
- video
|
|
- render
|
|
security_opt:
|
|
- seccomp=unconfined
|
|
ipc: host
|
|
volumes:
|
|
- models:/models
|
|
command: >
|
|
-m /models/${EMBEDDING_MODEL_FILE:-nomic-embed-text-v1.5.Q8_0.gguf}
|
|
--host 0.0.0.0
|
|
--port 8080
|
|
--embeddings
|
|
--pooling mean
|
|
--n-gpu-layers 999
|
|
--ctx-size 8192
|
|
# A dedicated embedding model — the chat model isn't embedding-trained
|
|
# and llama.cpp serves one model per process, so this is a second small
|
|
# instance, not a mode switch on llama-server. See
|
|
# docs/research/litellm-knowledgebase.md. Small enough (~150MB Q8) to
|
|
# run alongside the chat model's ~19.6GB in the R9700's 32GB VRAM.
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
|
|
# ponytail: one-off downloader, not a standing service — run via
|
|
# `docker compose --profile tools run --rm downloader` (see scripts/download-model.sh).
|
|
# Keeps the model file inside the named `models` volume instead of a host bind-mount.
|
|
downloader:
|
|
image: curlimages/curl:latest
|
|
profiles: ["tools"]
|
|
# ponytail: named volume is created root-owned; curl_user (uid 100) can't
|
|
# write into it otherwise, so run as root for this one-off job.
|
|
user: root
|
|
volumes:
|
|
- models:/models
|
|
entrypoint: ["sh", "-c"]
|
|
command:
|
|
- >
|
|
curl -L --fail --create-dirs -o /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
|
|
https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
|
|
|
|
# ponytail: same one-off pattern as `downloader`, for the embedding model —
|
|
# run via `docker compose --profile tools run --rm downloader-embedding`.
|
|
downloader-embedding:
|
|
image: curlimages/curl:latest
|
|
profiles: ["tools"]
|
|
user: root
|
|
volumes:
|
|
- models:/models
|
|
entrypoint: ["sh", "-c"]
|
|
command:
|
|
- >
|
|
curl -L --fail --create-dirs -o /models/${EMBEDDING_MODEL_FILE:-nomic-embed-text-v1.5.Q8_0.gguf}
|
|
https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/${EMBEDDING_MODEL_FILE:-nomic-embed-text-v1.5.Q8_0.gguf}
|
|
|
|
qdrant:
|
|
image: qdrant/qdrant:latest
|
|
container_name: qdrant
|
|
volumes:
|
|
- qdrant-data:/qdrant/storage
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/6333'"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
retries: 5
|
|
|
|
open-webui:
|
|
image: ghcr.io/open-webui/open-webui:main
|
|
container_name: open-webui
|
|
depends_on:
|
|
qdrant:
|
|
condition: service_healthy
|
|
litellm:
|
|
condition: service_healthy
|
|
volumes:
|
|
- openwebui-data:/app/backend/data
|
|
env_file: .env
|
|
environment:
|
|
- WEBUI_AUTH=True
|
|
# Routed through the litellm proxy, not llama-server directly — see issue #15.
|
|
# OPENAI_API_KEY must be a virtual key created for Open WebUI per
|
|
# docs/proxy-key-onboarding.md (name it "openwebui"), set as
|
|
# OPENWEBUI_LITELLM_KEY in .env.
|
|
- OPENAI_API_BASE_URL=http://litellm:4000/v1
|
|
- OPENAI_API_KEY=${OPENWEBUI_LITELLM_KEY}
|
|
- VECTOR_DB=qdrant
|
|
- QDRANT_URI=http://qdrant:6333
|
|
ports:
|
|
- "${WEBUI_PORT:-8008}:8080"
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
|
|
litellm:
|
|
image: ghcr.io/berriai/litellm:main-stable
|
|
container_name: litellm
|
|
depends_on:
|
|
litellm-db:
|
|
condition: service_healthy
|
|
llama-server:
|
|
condition: service_started
|
|
redis:
|
|
condition: service_healthy
|
|
volumes:
|
|
- ./litellm-config.yaml:/app/config.yaml:ro
|
|
# LITELLM_MASTER_KEY / LITELLM_SALT_KEY come straight from .env via env_file
|
|
# (names match what litellm reads). LITELLM_SALT_KEY must not change after
|
|
# first run — see .env.example.
|
|
env_file: .env
|
|
environment:
|
|
- DATABASE_URL=postgresql://litellm:${LITELLM_DB_PASSWORD}@litellm-db:5432/litellm
|
|
# Setting these is all LiteLLM needs to use Redis for router state,
|
|
# rate limits/budgets, and cache invalidation — no extra config.yaml
|
|
# block required. See https://docs.litellm.ai/docs/proxy/caching.
|
|
- REDIS_HOST=redis
|
|
- REDIS_PORT=6379
|
|
- REDIS_PASSWORD=${REDIS_PASSWORD}
|
|
# The litellm container only joins the ai-stack bridge network, which has
|
|
# no visibility into the LAN's mDNS/local-DNS names — search.home won't
|
|
# resolve without this. Set SEARXNG_LAN_IP in .env to its stable LAN IP
|
|
# (static DHCP reservation recommended). See docs/research/litellm-searxng-search.md.
|
|
extra_hosts:
|
|
- "search.home:${SEARXNG_LAN_IP}"
|
|
command: ["--config", "/app/config.yaml", "--port", "4000"]
|
|
ports:
|
|
# published for LAN access (proxy.ai.home) and, via NPM, proxy.ai.haylan.ch —
|
|
# NPM must deny the /ui path on the external host. See docs/network-access.md.
|
|
- "${LITELLM_PORT:-4000}:4000"
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
healthcheck:
|
|
test:
|
|
- CMD-SHELL
|
|
- python3 -c "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"
|
|
interval: 30s
|
|
timeout: 10s
|
|
retries: 3
|
|
start_period: 40s
|
|
|
|
litellm-db:
|
|
image: postgres:16-alpine
|
|
container_name: litellm-db
|
|
env_file: .env
|
|
environment:
|
|
- POSTGRES_USER=litellm
|
|
- POSTGRES_PASSWORD=${LITELLM_DB_PASSWORD}
|
|
- POSTGRES_DB=litellm
|
|
volumes:
|
|
- litellm-db-data:/var/lib/postgresql/data
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "pg_isready -d litellm -U litellm"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 10
|
|
|
|
# Backs litellm's router state, rate limits/budgets, and cache
|
|
# invalidation (see the litellm service's REDIS_* env vars above).
|
|
# ponytail: no persistence volume — everything litellm stores here is
|
|
# cache/coordination state it's fine to lose on restart, not source data.
|
|
redis:
|
|
image: redis:7-alpine
|
|
container_name: redis
|
|
command: ["redis-server", "--requirepass", "${REDIS_PASSWORD}"]
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "redis-cli -a ${REDIS_PASSWORD} ping | grep -q PONG"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 10
|
|
|
|
# Separate Postgres instance (with the pgvector extension) for the
|
|
# knowledgebase — NOT the same database as litellm-db, which is plain
|
|
# postgres:16-alpine and has no vector extension installed. See
|
|
# docs/research/litellm-knowledgebase.md.
|
|
pgvector-db:
|
|
image: pgvector/pgvector:pg16
|
|
container_name: pgvector-db
|
|
env_file: .env
|
|
environment:
|
|
- POSTGRES_USER=litellm_pgvector
|
|
- POSTGRES_PASSWORD=${PGVECTOR_DB_PASSWORD}
|
|
- POSTGRES_DB=litellm_pgvector
|
|
volumes:
|
|
- pgvector-db-data:/var/lib/postgresql/data
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "pg_isready -d litellm_pgvector -U litellm_pgvector"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 10
|
|
|
|
# LiteLLM's native knowledgebase/vector-store feature has no Qdrant backend
|
|
# (the qdrant service above only serves Open WebUI's own RAG/Memory) — this
|
|
# companion service (github.com/BerriAI/litellm-pgvector) is the only
|
|
# self-hosted path. No published image exists yet, so this builds from a
|
|
# vendored copy in vendor/litellm-pgvector/ (see that dir's README) rather
|
|
# than a remote git build context — the server's Docker/BuildKit couldn't
|
|
# do an authenticated-looking clone of a public github.com repo (fails
|
|
# with "could not read Username ... terminal prompts disabled"), and
|
|
# vendoring sidesteps needing that debugged. See
|
|
# docs/research/litellm-knowledgebase.md.
|
|
# ponytail: unverified against real hardware — Prisma migration behavior on
|
|
# first boot and the exact vector_store_registry field names for the
|
|
# pg_vector provider need a live smoke test. See issue #24.
|
|
litellm-pgvector:
|
|
build:
|
|
context: ./vendor/litellm-pgvector
|
|
container_name: litellm-pgvector
|
|
depends_on:
|
|
pgvector-db:
|
|
condition: service_healthy
|
|
litellm:
|
|
condition: service_healthy
|
|
environment:
|
|
- DATABASE_URL=postgresql://litellm_pgvector:${PGVECTOR_DB_PASSWORD}@pgvector-db:5432/litellm_pgvector
|
|
- SERVER_API_KEY=${LITELLM_PGVECTOR_API_KEY}
|
|
# Calls back into litellm for embeddings, same pattern as any other
|
|
# workload — see docs/proxy-key-onboarding.md for issuing this key.
|
|
- EMBEDDING__MODEL=local-embedding
|
|
- EMBEDDING__BASE_URL=http://litellm:4000
|
|
- EMBEDDING__API_KEY=${LITELLM_PGVECTOR_EMBEDDING_KEY}
|
|
- EMBEDDING__DIMENSIONS=768
|
|
expose:
|
|
- "8000"
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
|
|
lazytainer:
|
|
image: ghcr.io/vmorganp/lazytainer:master
|
|
container_name: lazytainer
|
|
# NOT network_mode: host — lazytainer identifies its own container by
|
|
# matching os.Hostname() against the Docker container-ID list
|
|
# (vmorganp/Lazytainer, configureFromLabels()); under host networking the
|
|
# container inherits the host's hostname instead of its own ID, so that
|
|
# match always fails and it panics with "Could not determine container ID
|
|
# of lazytainer" on every start. Host networking also can't see traffic
|
|
# to llama-server:8080 anyway — that port only exists on the ai-stack
|
|
# bridge network (no host port published, see issue #15 above). Joining
|
|
# ai-stack instead fixes both: hostname becomes the real container ID,
|
|
# and it's on the same network as the traffic it's watching.
|
|
networks: [ai-stack]
|
|
volumes:
|
|
- /var/run/docker.sock:/var/run/docker.sock:ro
|
|
restart: unless-stopped
|
|
depends_on:
|
|
- llama-server
|
|
|
|
networks:
|
|
ai-stack:
|
|
|
|
volumes:
|
|
models:
|
|
qdrant-data:
|
|
openwebui-data:
|
|
litellm-db-data:
|
|
pgvector-db-data:
|