diff --git a/config.yaml b/config.yaml index 4c46aa8..0e3992f 100644 --- a/config.yaml +++ b/config.yaml @@ -4,6 +4,13 @@ # which verifies the enclave attestation and encrypts request/response bodies # with EHBP (HPKE) before forwarding to the Tinfoil enclave. Provider key set # via environment variable TINFOIL_API_KEY. +# +# Pricing (model_info) is per-token, derived from Tinfoil's per-1M rates: +# deepseek-v4-1-flash: $0.65 in / $1.45 out / $0.13 cached per 1M +# gpt-oss-120b: $0.15 in / $0.60 out per 1M +# glm-5-3-flash: $0.40 in / $1.25 out / $0.10 cached per 1M +# Without these, LiteLLM can't price these custom models, so spend shows $0.00 +# and the team budget cap is not enforced. model_list: # DeepSeek V4.1 Flash — default model (replaces V4 Flash, deprecated 2026-09-15) @@ -12,6 +19,10 @@ model_list: model: openai/deepseek-v4-1-flash api_base: http://127.0.0.1:3301/v1 api_key: os.environ/TINFOIL_API_KEY + model_info: + input_cost_per_token: 0.00000065 + output_cost_per_token: 0.00000145 + cache_read_input_token_cost: 0.00000013 # GPT-OSS 120B — lightweight fallback - model_name: gpt-oss-120b @@ -19,6 +30,9 @@ model_list: model: openai/gpt-oss-120b api_base: http://127.0.0.1:3301/v1 api_key: os.environ/TINFOIL_API_KEY + model_info: + input_cost_per_token: 0.00000015 + output_cost_per_token: 0.00000060 # GLM-5.3 Flash — fast, efficient MoE model - model_name: glm-5-3-flash @@ -26,6 +40,10 @@ model_list: model: openai/glm-5-3-flash api_base: http://127.0.0.1:3301/v1 api_key: os.environ/TINFOIL_API_KEY + model_info: + input_cost_per_token: 0.00000040 + output_cost_per_token: 0.00000125 + cache_read_input_token_cost: 0.00000010 # Fallback: if DeepSeek is down, use GPT-OSS router_settings: