From 576885ed2138adfc1c748df68088c07f13013dfd Mon Sep 17 00:00:00 2001 From: inference-bot Date: Tue, 22 Sep 2026 18:05:27 -0600 Subject: [PATCH] Fix Tinfoil spend gap: raise timeout 30->180s, num_retries 2->0 DeepSeek V4.1 Flash is a reasoning model; 30s timeout caused LiteLLM to abandon in-flight generations (falling back to GPT-OSS) while Tinfoil still completed and billed them. num_retries re-hit Tinfoil up to 3x. Result: Tinfoil billed ~2x the tokens LiteLLM logged. Generous timeout + zero retries stops the silent re-billing. --- config.yaml | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/config.yaml b/config.yaml index 0e3992f..94ba00a 100644 --- a/config.yaml +++ b/config.yaml @@ -46,9 +46,16 @@ model_list: cache_read_input_token_cost: 0.00000010 # Fallback: if DeepSeek is down, use GPT-OSS +# NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run +# longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek +# requests (falling back to GPT-OSS), while Tinfoil still completed and billed +# the abandoned generation. num_retries amplified this by re-hitting Tinfoil up +# to 3x per request. Result: Tinfoil billed ~2x the tokens LiteLLM logged. +# Fix: a generous timeout so generations complete inside LiteLLM (and get +# logged), and zero retries so a slow/failed request is never re-billed. router_settings: - num_retries: 2 - timeout: 30 + num_retries: 0 + timeout: 180 fallbacks: - deepseek-v4-1-flash: - gpt-oss-120b @@ -62,4 +69,4 @@ litellm_settings: salt_key: os.environ/LITELLM_SALT_KEY drop_params: true num_threads: 4 - request_timeout: 30 + request_timeout: 180