From d1f57c45d53ec16b0e5fae5f05047cb08f2fc197 Mon Sep 17 00:00:00 2001 From: inference-bot Date: Tue, 22 Sep 2026 18:18:07 -0600 Subject: [PATCH] Retry connection errors but never re-bill timeouts num_retries=2 (retries transient connection errors), retry_policy with TimeoutErrorRetries=0 (a timed-out generation is already billed by Tinfoil, so re-hitting it re-bills). Combined with timeout=180, this hedges connection errors without the silent re-billing that caused the Tinfoil spend gap. --- config.yaml | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/config.yaml b/config.yaml index 94ba00a..8cfba99 100644 --- a/config.yaml +++ b/config.yaml @@ -49,13 +49,16 @@ model_list: # NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run # longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek # requests (falling back to GPT-OSS), while Tinfoil still completed and billed -# the abandoned generation. num_retries amplified this by re-hitting Tinfoil up -# to 3x per request. Result: Tinfoil billed ~2x the tokens LiteLLM logged. -# Fix: a generous timeout so generations complete inside LiteLLM (and get -# logged), and zero retries so a slow/failed request is never re-billed. +# the abandoned generation. Result: Tinfoil billed ~2x the tokens LiteLLM logged. +# Fix: a generous timeout (180s) so generations complete inside LiteLLM and get +# logged; and a retry_policy that retries transient connection errors (base +# num_retries) but NEVER retries timeouts (TimeoutErrorRetries: 0) — a timed-out +# generation is already billed by Tinfoil, so re-hitting it re-bills. router_settings: - num_retries: 0 + num_retries: 2 timeout: 180 + retry_policy: + TimeoutErrorRetries: 0 fallbacks: - deepseek-v4-1-flash: - gpt-oss-120b