# LiteLLM config for inference.coop # PRODUCTION config: Tinfoil (TEE-protected inference) # LiteLLM talks plaintext OpenAI to a local Tinfoil proxy (127.0.0.1:3301), # which verifies the enclave attestation and encrypts request/response bodies # with EHBP (HPKE) before forwarding to the Tinfoil enclave. Provider key set # via environment variable TINFOIL_API_KEY. # # Pricing (model_info) is per-token, derived from Tinfoil's per-1M rates: # deepseek-v4-1-flash: $0.65 in / $1.45 out / $0.13 cached per 1M # gpt-oss-120b: $0.15 in / $0.60 out per 1M # glm-5-3-flash: $0.40 in / $1.25 out / $0.10 cached per 1M # Without these, LiteLLM can't price these custom models, so spend shows $0.00 # and the team budget cap is not enforced. model_list: # DeepSeek V4.1 Flash — default model (replaces V4 Flash, deprecated 2026-09-15) - model_name: deepseek-v4-1-flash litellm_params: model: openai/deepseek-v4-1-flash api_base: http://127.0.0.1:3301/v1 api_key: os.environ/TINFOIL_API_KEY model_info: input_cost_per_token: 0.00000065 output_cost_per_token: 0.00000145 cache_read_input_token_cost: 0.00000013 # GPT-OSS 120B — lightweight fallback - model_name: gpt-oss-120b litellm_params: model: openai/gpt-oss-120b api_base: http://127.0.0.1:3301/v1 api_key: os.environ/TINFOIL_API_KEY model_info: input_cost_per_token: 0.00000015 output_cost_per_token: 0.00000060 # GLM-5.3 Flash — fast, efficient MoE model - model_name: glm-5-3-flash litellm_params: model: openai/glm-5-3-flash api_base: http://127.0.0.1:3301/v1 api_key: os.environ/TINFOIL_API_KEY model_info: input_cost_per_token: 0.00000040 output_cost_per_token: 0.00000125 cache_read_input_token_cost: 0.00000010 # Fallback: if DeepSeek is down, use GPT-OSS # NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run # longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek # requests (falling back to GPT-OSS), while Tinfoil still completed and billed # the abandoned generation. num_retries amplified this by re-hitting Tinfoil up # to 3x per request. Result: Tinfoil billed ~2x the tokens LiteLLM logged. # Fix: a generous timeout so generations complete inside LiteLLM (and get # logged), and zero retries so a slow/failed request is never re-billed. router_settings: num_retries: 0 timeout: 180 fallbacks: - deepseek-v4-1-flash: - gpt-oss-120b general_settings: master_key: os.environ/LITELLM_MASTER_KEY database_url: os.environ/DATABASE_URL store_model_in_db: true litellm_settings: salt_key: os.environ/LITELLM_SALT_KEY drop_params: true num_threads: 4 request_timeout: 180