From 2369c3e848756e90895c1e6c0d543e739c3a6d37 Mon Sep 17 00:00:00 2001 From: inference-bot Date: Thu, 24 Sep 2026 10:43:14 -0600 Subject: [PATCH] Add greenpt/glm-5.3-flash (/usr/bin/bash.11//usr/bin/bash.44 per 1M) --- config.yaml | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/config.yaml b/config.yaml index b74354a..9280c5f 100644 --- a/config.yaml +++ b/config.yaml @@ -73,6 +73,18 @@ model_list: input_cost_per_token: 0.00000025 output_cost_per_token: 0.00000080 + # GLM-5.3 Flash on GreenPT (note the dotted slug — distinct from Tinfoil's + # glm-5-3-flash). $0.11 in / $0.44 out / $0.022 cached per 1M. + - model_name: greenpt/glm-5.3-flash + litellm_params: + model: openai/glm-5.3-flash + api_base: https://api.greenpt.ai/v1 + api_key: os.environ/GREENPT_API_KEY + model_info: + input_cost_per_token: 0.00000011 + output_cost_per_token: 0.00000044 + cache_read_input_token_cost: 0.000000022 + # Fallback: if DeepSeek is down, use GPT-OSS # NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run # longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek