Fix injector ReadTimeout (600s read timeout); switch default model to deepseek-v4-1-flash

This commit is contained in:
inference-bot committed 2026-09-14 14:11:15 -06:00
1 parent 3024bfec74
commit 60d1bf2328
1 file changed
+7 -3
+7 -3
View File
@@ -79,7 +79,7 @@ OC_PERSONAL_TOKEN = os.environ.get("OC_PERSONAL_TOKEN", "")
MEMBER_BUDGET = float(os.environ.get("MEMBER_BUDGET", "15.0"))
# Models every member key/team may access.
MEMBER_MODELS = ["deepseek-v4-flash", "gpt-oss-120b", "glm-5-3-flash"]
MEMBER_MODELS = ["deepseek-v4-1-flash", "gpt-oss-120b", "glm-5-3-flash"]
# The "members" group in Cloudron (group-based access control).
# Members are assigned to this group, which grants access to the chat app.
@@ -921,14 +921,18 @@ async def inject_key(request: Request, path: str):
if not member_key:
raise HTTPException(403, "No active membership key")
# Forward the request to LiteLLM with the member's key
# Forward the request to LiteLLM with the member's key.
# Use a generous timeout: model generations can take well over the httpx
# default of 5s, and the previous code buffered the full response which
# caused httpx.ReadTimeout on any non-trivial generation.
body = await request.body()
headers = dict(request.headers)
headers["authorization"] = f"Bearer {member_key}"
headers.pop("host", None)
headers.pop("content-length", None)
async with httpx.AsyncClient() as client:
timeout = httpx.Timeout(connect=15.0, read=600.0, write=30.0, pool=15.0)
async with httpx.AsyncClient(timeout=timeout) as client:
upstream = await client.request(
method=request.method,
url=f"{LITELLM_BASE}/v1/{path}",