diff --git a/config.yaml b/config.yaml index 94ba00a..8cfba99 100644 --- a/config.yaml +++ b/config.yaml @@ -49,13 +49,16 @@ model_list: # NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run # longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek # requests (falling back to GPT-OSS), while Tinfoil still completed and billed -# the abandoned generation. num_retries amplified this by re-hitting Tinfoil up -# to 3x per request. Result: Tinfoil billed ~2x the tokens LiteLLM logged. -# Fix: a generous timeout so generations complete inside LiteLLM (and get -# logged), and zero retries so a slow/failed request is never re-billed. +# the abandoned generation. Result: Tinfoil billed ~2x the tokens LiteLLM logged. +# Fix: a generous timeout (180s) so generations complete inside LiteLLM and get +# logged; and a retry_policy that retries transient connection errors (base +# num_retries) but NEVER retries timeouts (TimeoutErrorRetries: 0) — a timed-out +# generation is already billed by Tinfoil, so re-hitting it re-bills. router_settings: - num_retries: 0 + num_retries: 2 timeout: 180 + retry_policy: + TimeoutErrorRetries: 0 fallbacks: - deepseek-v4-1-flash: - gpt-oss-120b