From 60d1bf23280589e9900c49384209357539a8d3c5 Mon Sep 17 00:00:00 2001 From: inference-bot Date: Mon, 14 Sep 2026 14:11:15 -0600 Subject: [PATCH] Fix injector ReadTimeout (600s read timeout); switch default model to deepseek-v4-1-flash --- app/main.py | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/app/main.py b/app/main.py index f37604a..a8571c5 100644 --- a/app/main.py +++ b/app/main.py @@ -79,7 +79,7 @@ OC_PERSONAL_TOKEN = os.environ.get("OC_PERSONAL_TOKEN", "") MEMBER_BUDGET = float(os.environ.get("MEMBER_BUDGET", "15.0")) # Models every member key/team may access. -MEMBER_MODELS = ["deepseek-v4-flash", "gpt-oss-120b", "glm-5-3-flash"] +MEMBER_MODELS = ["deepseek-v4-1-flash", "gpt-oss-120b", "glm-5-3-flash"] # The "members" group in Cloudron (group-based access control). # Members are assigned to this group, which grants access to the chat app. @@ -921,14 +921,18 @@ async def inject_key(request: Request, path: str): if not member_key: raise HTTPException(403, "No active membership key") - # Forward the request to LiteLLM with the member's key + # Forward the request to LiteLLM with the member's key. + # Use a generous timeout: model generations can take well over the httpx + # default of 5s, and the previous code buffered the full response which + # caused httpx.ReadTimeout on any non-trivial generation. body = await request.body() headers = dict(request.headers) headers["authorization"] = f"Bearer {member_key}" headers.pop("host", None) headers.pop("content-length", None) - async with httpx.AsyncClient() as client: + timeout = httpx.Timeout(connect=15.0, read=600.0, write=30.0, pool=15.0) + async with httpx.AsyncClient(timeout=timeout) as client: upstream = await client.request( method=request.method, url=f"{LITELLM_BASE}/v1/{path}",