LiteLLM -> OmniRoute (issue #31, wayfinder map + research tickets #32-37): replace the litellm/litellm-db services with omniroute, split-port mode (API_PORT published/reverse-proxied, DASHBOARD_PORT never published - tighter than litellm's old /ui NPM path-deny rule), 5 new secrets in place of LITELLM_MASTER_KEY/LITELLM_SALT_KEY, llama-server/searxng registered as omniroute providers post-boot (no static config.yaml equivalent). No scripted per-workload key minting yet - omniroute's POST /api/keys needs a dashboard session, not a static bearer key - so OPENWEBUI_OMNIROUTE_KEY is a manual step for now (docs/proxy-key-onboarding.md). Caveat carried into the map and README: OmniRoute's own docs (docs/security/STEALTH_GUIDE.md, MITM-TPROXY-DECRYPT.md, PUBLIC_CREDS.md on its release/v3.8.51 branch) describe shipped features for AI-provider client-detection evasion, system-wide HTTPS interception via a locally installed root CA, and hiding credentials from secret scanners. Proceeding anyway was an explicit, informed user decision. Also drops the gateway-level memory/knowledgebase feature entirely (user: "I don't need it") - litellm-pgvector, pgvector-db, embedding-server, scripts/ingest-memory.sh, vendor/litellm-pgvector/, docs/memory- knowledgebase.md. Open WebUI's own qdrant-backed memory/RAG is unrelated and untouched. litellm-config.yaml deleted (was kept as a rollback reference, but there's no rollback path to a feature being deliberately removed). Not yet verified against real hardware - see issue #31's open tickets. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01VPZ6TogJiYxG8E4EQBB197
171 lines
6.4 KiB
YAML
171 lines
6.4 KiB
YAML
services:
|
|
llama-server:
|
|
image: ghcr.io/ggml-org/llama.cpp:server-rocm
|
|
container_name: llama-server
|
|
devices:
|
|
- /dev/kfd
|
|
- /dev/dri
|
|
group_add:
|
|
- video
|
|
- render
|
|
security_opt:
|
|
- seccomp=unconfined
|
|
ipc: host
|
|
volumes:
|
|
- models:/models
|
|
command: >
|
|
-m /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
|
|
--host 0.0.0.0
|
|
--port 8080
|
|
--n-gpu-layers ${LLAMA_GPU_LAYERS:-999}
|
|
--ctx-size ${LLAMA_CTX_SIZE:-131072}
|
|
--jinja
|
|
# No published host port: llama-server is reached only via the litellm
|
|
# proxy on the ai-stack docker network now — see issue #15. Its
|
|
# unauthenticated API no longer needs to be LAN-reachable directly.
|
|
expose:
|
|
- "8080"
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
labels:
|
|
# ponytail: idle-timeout tuning lives here, not in a separate lazytainer config file —
|
|
# one place to look. Raise LAZYTAINER_INACTIVE_TIMEOUT if 15 min proves too eager.
|
|
- "lazytainer.group.llamaserver.sleepMethod=stop"
|
|
- "lazytainer.group.llamaserver.ports=8080"
|
|
- "lazytainer.group.llamaserver.inactiveTimeout=${LAZYTAINER_INACTIVE_TIMEOUT:-900}"
|
|
- "lazytainer.group.llamaserver.minPacketThreshold=2"
|
|
|
|
# ponytail: one-off downloader, not a standing service — run via
|
|
# `docker compose --profile tools run --rm downloader`. Folded into
|
|
# scripts/update.sh, which runs this every time; the `test -f` guard is
|
|
# what makes that safe to re-run without re-downloading. Keeps the model
|
|
# file inside the named `models` volume instead of a host bind-mount.
|
|
downloader:
|
|
image: curlimages/curl:latest
|
|
profiles: ["tools"]
|
|
# ponytail: named volume is created root-owned; curl_user (uid 100) can't
|
|
# write into it otherwise, so run as root for this one-off job.
|
|
user: root
|
|
volumes:
|
|
- models:/models
|
|
entrypoint: ["sh", "-c"]
|
|
command:
|
|
- >
|
|
test -f /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf} &&
|
|
echo "already downloaded, skipping" ||
|
|
curl -L --fail --create-dirs -o /models/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
|
|
https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/${LLAMA_MODEL_FILE:-Qwen3.8-27B-UD-Q4_K_XL.gguf}
|
|
|
|
qdrant:
|
|
image: qdrant/qdrant:latest
|
|
container_name: qdrant
|
|
volumes:
|
|
- qdrant-data:/qdrant/storage
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "bash -c 'exec 3<>/dev/tcp/localhost/6333'"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
retries: 5
|
|
|
|
open-webui:
|
|
image: ghcr.io/open-webui/open-webui:main
|
|
container_name: open-webui
|
|
depends_on:
|
|
qdrant:
|
|
condition: service_healthy
|
|
omniroute:
|
|
condition: service_healthy
|
|
volumes:
|
|
- openwebui-data:/app/backend/data
|
|
env_file: .env
|
|
environment:
|
|
- WEBUI_AUTH=True
|
|
# Routed through the omniroute gateway, not llama-server directly — see
|
|
# issue #15 (original rationale) and #31 (litellm -> omniroute
|
|
# migration). OPENAI_API_KEY must be a per-workload key created for
|
|
# Open WebUI in the omniroute dashboard (Keys -> Create, label
|
|
# "openwebui") — no scripted mint yet, see docs/proxy-key-onboarding.md.
|
|
- OPENAI_API_BASE_URL=http://omniroute:${OMNIROUTE_API_PORT:-20129}/v1
|
|
- OPENAI_API_KEY=${OPENWEBUI_OMNIROUTE_KEY}
|
|
- VECTOR_DB=qdrant
|
|
- QDRANT_URI=http://qdrant:6333
|
|
ports:
|
|
- "${WEBUI_PORT:-8008}:8080"
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
|
|
# Replaces litellm — see issue #31 (wayfinder map) for the full migration
|
|
# rationale/findings. No static config.yaml equivalent: provider routing
|
|
# (llama-server, searxng-search) is registered once through the dashboard
|
|
# or POST /api/providers after first boot, not checked into this repo —
|
|
# see docs/proxy-key-onboarding.md.
|
|
omniroute:
|
|
image: diegosouzapw/omniroute:latest
|
|
container_name: omniroute
|
|
depends_on:
|
|
llama-server:
|
|
condition: service_started
|
|
volumes:
|
|
- omniroute-data:/app/data
|
|
env_file: .env
|
|
environment:
|
|
# Split-port mode: dashboard and API are fully separate ports (unlike
|
|
# LiteLLM's single :4000 for both /v1 and /ui) — only API_PORT is
|
|
# published below, so the dashboard has no network route in from
|
|
# outside this container at all. No NPM path-deny rule needed.
|
|
- API_HOST=0.0.0.0
|
|
- API_PORT=${OMNIROUTE_API_PORT:-20129}
|
|
- DASHBOARD_PORT=${OMNIROUTE_DASHBOARD_PORT:-20128}
|
|
# Required to register llama-server/searxng-search as providers —
|
|
# their base URLs are LAN/container-internal addresses, blocked by
|
|
# default (SSRF guard against public-provider spoofing).
|
|
- OMNIROUTE_ALLOW_PRIVATE_PROVIDER_URLS=true
|
|
- OMNIROUTE_ALLOW_LOCAL_PROVIDER_URLS=true
|
|
# Same reasoning as litellm's extra_hosts entry below — ai-stack's bridge
|
|
# network can't resolve search.home on its own.
|
|
extra_hosts:
|
|
- "search.home:${SEARXNG_LAN_IP}"
|
|
ports:
|
|
- "${OMNIROUTE_API_PORT:-20129}:${OMNIROUTE_API_PORT:-20129}"
|
|
restart: unless-stopped
|
|
networks: [ai-stack]
|
|
healthcheck:
|
|
test:
|
|
- CMD-SHELL
|
|
- python3 -c "import urllib.request; urllib.request.urlopen('http://localhost:${OMNIROUTE_API_PORT:-20129}/healthz')"
|
|
interval: 30s
|
|
timeout: 10s
|
|
retries: 3
|
|
start_period: 40s
|
|
|
|
lazytainer:
|
|
image: ghcr.io/vmorganp/lazytainer:master
|
|
container_name: lazytainer
|
|
# NOT network_mode: host — lazytainer identifies its own container by
|
|
# matching os.Hostname() against the Docker container-ID list
|
|
# (vmorganp/Lazytainer, configureFromLabels()); under host networking the
|
|
# container inherits the host's hostname instead of its own ID, so that
|
|
# match always fails and it panics with "Could not determine container ID
|
|
# of lazytainer" on every start. Host networking also can't see traffic
|
|
# to llama-server:8080 anyway — that port only exists on the ai-stack
|
|
# bridge network (no host port published, see issue #15 above). Joining
|
|
# ai-stack instead fixes both: hostname becomes the real container ID,
|
|
# and it's on the same network as the traffic it's watching.
|
|
networks: [ai-stack]
|
|
volumes:
|
|
- /var/run/docker.sock:/var/run/docker.sock:ro
|
|
restart: unless-stopped
|
|
depends_on:
|
|
- llama-server
|
|
|
|
networks:
|
|
ai-stack:
|
|
|
|
volumes:
|
|
models:
|
|
qdrant-data:
|
|
openwebui-data:
|
|
omniroute-data:
|