Add custom per-token pricing (model_info) so spend and budget enforcement work

This commit is contained in:
inference-bot committed 2026-09-14 14:28:13 -06:00
1 parent da50b0438f
commit 700154c3d4
1 file changed
+18
+18
View File
@@ -4,6 +4,13 @@
# which verifies the enclave attestation and encrypts request/response bodies # which verifies the enclave attestation and encrypts request/response bodies
# with EHBP (HPKE) before forwarding to the Tinfoil enclave. Provider key set # with EHBP (HPKE) before forwarding to the Tinfoil enclave. Provider key set
# via environment variable TINFOIL_API_KEY. # via environment variable TINFOIL_API_KEY.
#
# Pricing (model_info) is per-token, derived from Tinfoil's per-1M rates:
# deepseek-v4-1-flash: $0.65 in / $1.45 out / $0.13 cached per 1M
# gpt-oss-120b: $0.15 in / $0.60 out per 1M
# glm-5-3-flash: $0.40 in / $1.25 out / $0.10 cached per 1M
# Without these, LiteLLM can't price these custom models, so spend shows $0.00
# and the team budget cap is not enforced.
model_list: model_list:
# DeepSeek V4.1 Flash — default model (replaces V4 Flash, deprecated 2026-09-15) # DeepSeek V4.1 Flash — default model (replaces V4 Flash, deprecated 2026-09-15)
@@ -12,6 +19,10 @@ model_list:
model: openai/deepseek-v4-1-flash model: openai/deepseek-v4-1-flash
api_base: http://127.0.0.1:3301/v1 api_base: http://127.0.0.1:3301/v1
api_key: os.environ/TINFOIL_API_KEY api_key: os.environ/TINFOIL_API_KEY
model_info:
input_cost_per_token: 0.00000065
output_cost_per_token: 0.00000145
cache_read_input_token_cost: 0.00000013
# GPT-OSS 120B — lightweight fallback # GPT-OSS 120B — lightweight fallback
- model_name: gpt-oss-120b - model_name: gpt-oss-120b
@@ -19,6 +30,9 @@ model_list:
model: openai/gpt-oss-120b model: openai/gpt-oss-120b
api_base: http://127.0.0.1:3301/v1 api_base: http://127.0.0.1:3301/v1
api_key: os.environ/TINFOIL_API_KEY api_key: os.environ/TINFOIL_API_KEY
model_info:
input_cost_per_token: 0.00000015
output_cost_per_token: 0.00000060
# GLM-5.3 Flash — fast, efficient MoE model # GLM-5.3 Flash — fast, efficient MoE model
- model_name: glm-5-3-flash - model_name: glm-5-3-flash
@@ -26,6 +40,10 @@ model_list:
model: openai/glm-5-3-flash model: openai/glm-5-3-flash
api_base: http://127.0.0.1:3301/v1 api_base: http://127.0.0.1:3301/v1
api_key: os.environ/TINFOIL_API_KEY api_key: os.environ/TINFOIL_API_KEY
model_info:
input_cost_per_token: 0.00000040
output_cost_per_token: 0.00000125
cache_read_input_token_cost: 0.00000010
# Fallback: if DeepSeek is down, use GPT-OSS # Fallback: if DeepSeek is down, use GPT-OSS
router_settings: router_settings: