# Copy to .env and adjust — or just run ./scripts/update.sh, which creates # .env from this file and fills in every secret/key below it can generate # itself (see each var's comment). All values below are defaults baked into # docker-compose.yml — only uncomment/change what you actually want to # override. # --- llama.cpp / model --- LLAMA_MODEL_FILE=Qwen3.8-27B-UD-Q4_K_XL.gguf # 999 = every layer on GPU (this model is dense, not MoE, and already fits # fully in 32GB VRAM — see docs/research/qwen3.8-27b-quant.md). Lower this # to leave that many fewer layers on GPU and push the rest to CPU/system RAM # if something else is contending for VRAM — llama.cpp has no separate # "RAM offload" flag, --n-gpu-layers *is* the RAM-offload knob for a dense # model. Don't reach for --n-cpu-moe/--cpu-moe/--override-tensor "exps" — # those target Mixture-of-Experts models (e.g. Qwen3.8-2.4T-A95B), not this # one, and are no-ops here. # There's no separate "then SSD" tier to enable either: llama.cpp mmaps the # model file by default (no --no-mmap here), so if GPU+RAM ever can't hold # the working set, the OS pages the rest in from disk automatically — an # implicit, slow last resort, not a config knob. An explicit tiered SSD # offload has been an open llama.cpp feature request since 2025 (still # unimplemented): https://github.com/ggml-org/llama.cpp/discussions/12507 LLAMA_GPU_LAYERS=999 # 262144 = this model's true max (max_position_embeddings in Qwen/Qwen3.8-27B's # config.json) — the largest --ctx-size llama.cpp will even accept for it. # fp16 KV cache at full context would be ~16GB, on top of 17.6GB weights = # ~33.6GB, which does NOT fit the 32GB R9700 on its own. docker-compose.yml # now runs --cache-type-k/v q8_0, which roughly halves KV memory (~8GB at # this size) — total ~25.6GB, ~6GB headroom, the same footprint the old # 131072 fp16 setting used. See docs/research/qwen3.8-27b-quant.md. LLAMA_CTX_SIZE=262144 # Concurrent request slots. Was implicitly 4 (llama.cpp's compiled-in # default) with no flag set — under concurrent subagent fan-out, 4 requests # split the same GPU compute, so a large-context prefill can queue behind # others long enough to blow past OmniRoute's stream-idle timeout, which then # cancels the request (see issue-tracker notes on the timeout/cancel loop). # Dropped to 2 so each slot gets more compute and finishes prefill sooner; # raise back toward 4 if throughput (not latency) becomes the bottleneck # instead. Each slot gets LLAMA_CTX_SIZE / LLAMA_PARALLEL tokens of context — # real sessions have hit ~66K tokens, so don't drop LLAMA_CTX_SIZE without # checking that per-slot number stays comfortably above observed usage. LLAMA_PARALLEL=2 # --- Lazytainer --- # Seconds of inactivity before llama-server is stopped. 900 = 15 min. LAZYTAINER_INACTIVE_TIMEOUT=900 # --- SearXNG web search (see docs/research/litellm-searxng-search.md) --- # Resolved automatically by ./scripts/update.sh from search.home on this # host — leave blank. Only set by hand if that resolution fails (e.g. # search.home isn't a static DHCP reservation and its IP drifted). SEARXNG_LAN_IP= # --- OmniRoute gateway (see docs/proxy-key-onboarding.md, docs/network-access.md) --- # OMNIROUTE_PORT is the host-published port (reverse-proxied by NPM) — kept # at 4000, same as the old LiteLLM setup, so existing NPM/firewall config # doesn't need to change. It's mapped via plain Docker port publishing onto # API_PORT, omniroute's own container-internal port (left at its default, # not reconfigured to match). The dashboard (DASHBOARD_PORT) is never # published at all — see docker-compose.yml's omniroute service comment. OMNIROUTE_API_PORT=20129 OMNIROUTE_DASHBOARD_PORT=20128 # SSE inactivity timeout before OmniRoute gives up on a streaming request and # cancels it (which cancels the matching llama-server task too). 180s gives # contended prefill (see LLAMA_PARALLEL above) room to produce a first token. OMNIROUTE_STREAM_IDLE_TIMEOUT_MS=180000 # Random values, filled in automatically by ./scripts/update.sh — leave # blank. Bootstrap dashboard admin password (log in at the dashboard port, # change it there afterwards — this is only the first-boot value): OMNIROUTE_INITIAL_PASSWORD= # Signs dashboard session cookies: OMNIROUTE_JWT_SECRET= # Encrypts API key values at rest in omniroute's SQLite DB: OMNIROUTE_API_KEY_SECRET= # Encrypts the whole SQLite DB at rest. Do not change after first run — # existing encrypted data becomes unreadable if you do (same caveat as # LiteLLM's old LITELLM_SALT_KEY): OMNIROUTE_STORAGE_ENCRYPTION_KEY= # Per-deployment salts — random is fine, just needs to be stable: OMNIROUTE_MACHINE_ID_SALT= OMNIROUTE_CLI_SALT= # Required (production) — shared secret for the internal Codex Responses # WebSocket bridge. Random value, filled in automatically: OMNIROUTE_WS_BRIDGE_SECRET= # Per-workload virtual keys (one per client that calls the gateway) have no # scripted /key/generate equivalent yet — omniroute's key-creation endpoint # needs a dashboard login session, not a static bearer key (see issue #37). # Mint them by hand in the dashboard, add a KEY=value line here per workload # as you onboard one. See docs/proxy-key-onboarding.md. # --- ComfyUI (local image generation, see issue #38 wayfinder map) --- # yurisasc/comfyui-rocm7.1 manages GPU-group access via these GID/UID env # vars rather than relying solely on docker-compose.yml's group_add. # Resolved automatically from the host by ./scripts/update.sh — leave blank. COMFYUI_PUID= COMFYUI_PGID= COMFYUI_VIDEO_GID= COMFYUI_RENDER_GID= # --- llama.cpp / fast model (second, always-resident instance — see # docs/research/fast-model-choice.md and issue #44) --- # Qwen3-4B-Instruct-2507: architecturally non-thinking (never emits # blocks, unlike Qwen3-1.7B/0.6B which need a per-call toggle) — # picked specifically so it stays fast enough for qwen-code's Auto Mode # classifier (Stage 1 wants ~300ms). Same publisher (unsloth) as the main # model for consistency. LLAMA_FAST_MODEL_FILE=Qwen3-4B-Instruct-2507-UD-Q8_K_XL.gguf # Same reasoning as LLAMA_GPU_LAYERS above — full GPU offload, this model # is dense too. LLAMA_FAST_GPU_LAYERS=999 # Classifier transcripts are truncated/bounded by qwen-code itself (see its # own Auto Mode docs) — no need for anywhere near the 27B's huge context. # 8192 keeps this instance's KV cache negligible. LLAMA_FAST_CTX_SIZE=8192 LLAMA_FAST_PARALLEL=2