diff --git a/config.yaml b/config.yaml index a62f9c6..b74354a 100644 --- a/config.yaml +++ b/config.yaml @@ -6,9 +6,9 @@ # via environment variable TINFOIL_API_KEY. # # Pricing (model_info) is per-token, derived from Tinfoil's per-1M rates: -# deepseek-v4-1-flash: $0.65 in / $1.45 out / $0.13 cached per 1M -# gpt-oss-120b: $0.15 in / $0.60 out per 1M -# glm-5-3-flash: $0.40 in / $1.25 out / $0.10 cached per 1M +# tinfoil/deepseek-v4-1-flash: $0.65 in / $1.45 out / $0.13 cached per 1M +# tinfoil/gpt-oss-120b: $0.15 in / $0.60 out per 1M +# tinfoil/glm-5-3-flash: $0.40 in / $1.25 out / $0.10 cached per 1M # Without these, LiteLLM can't price these custom models, so spend shows $0.00 # and the team budget cap is not enforced. # @@ -19,7 +19,7 @@ model_list: # DeepSeek V4.1 Flash — default model (replaces V4 Flash, deprecated 2026-09-15) - - model_name: deepseek-v4-1-flash + - model_name: tinfoil/deepseek-v4-1-flash litellm_params: model: openai/deepseek-v4-1-flash api_base: http://127.0.0.1:3301/v1 @@ -30,7 +30,7 @@ model_list: cache_read_input_token_cost: 0.00000013 # GPT-OSS 120B — lightweight fallback - - model_name: gpt-oss-120b + - model_name: tinfoil/gpt-oss-120b litellm_params: model: openai/gpt-oss-120b api_base: http://127.0.0.1:3301/v1 @@ -40,7 +40,7 @@ model_list: output_cost_per_token: 0.00000060 # GLM-5.3 Flash — fast, efficient MoE model - - model_name: glm-5-3-flash + - model_name: tinfoil/glm-5-3-flash litellm_params: model: openai/glm-5-3-flash api_base: http://127.0.0.1:3301/v1 @@ -88,8 +88,8 @@ router_settings: retry_policy: TimeoutErrorRetries: 0 fallbacks: - - deepseek-v4-1-flash: - - gpt-oss-120b + - tinfoil/deepseek-v4-1-flash: + - tinfoil/gpt-oss-120b general_settings: master_key: os.environ/LITELLM_MASTER_KEY