diff --git a/config.yaml b/config.yaml index b74354a..9280c5f 100644 --- a/config.yaml +++ b/config.yaml @@ -73,6 +73,18 @@ model_list: input_cost_per_token: 0.00000025 output_cost_per_token: 0.00000080 + # GLM-5.3 Flash on GreenPT (note the dotted slug — distinct from Tinfoil's + # glm-5-3-flash). $0.11 in / $0.44 out / $0.022 cached per 1M. + - model_name: greenpt/glm-5.3-flash + litellm_params: + model: openai/glm-5.3-flash + api_base: https://api.greenpt.ai/v1 + api_key: os.environ/GREENPT_API_KEY + model_info: + input_cost_per_token: 0.00000011 + output_cost_per_token: 0.00000044 + cache_read_input_token_cost: 0.000000022 + # Fallback: if DeepSeek is down, use GPT-OSS # NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run # longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek