Merge pull request 'fix(litellm): default max_tokens=4096 for the reasoning model' (#20) from fix/litellm-reasoning-max-tokens into main

This commit was merged in pull request #20.
This commit is contained in:
2026-09-02 18:58:11 +00:00
+7
View File
@@ -7,6 +7,13 @@ model_list:
model: openai/qwen3.8-27b-local
api_base: http://llama-server:8080/v1
api_key: local
# Qwen3 is a reasoning model — it spends output tokens on
# reasoning_content before ever writing content. Callers that don't
# set their own max_tokens (Open WebUI's default request didn't) hit
# llama.cpp's low default, so the model runs out mid-thought and
# content comes back empty. This is a floor, not a cap — any caller
# that passes its own max_tokens still overrides it.
max_tokens: 4096
model_info:
# Shadow cloud-cost estimate — priced against Claude Sonnet 5's published
# rate, not real spend (this proxy only ever routes to the local model).