From 057507d67c10e75c1a74a44a060348b363fbc17c Mon Sep 17 00:00:00 2001 From: inference-bot Date: Fri, 25 Sep 2026 15:56:08 -0600 Subject: [PATCH] Add PublicAI Apertus models (publicai/apertus-v1.5-8b, -70b) --- config.yaml | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/config.yaml b/config.yaml index 9280c5f..2bb405a 100644 --- a/config.yaml +++ b/config.yaml @@ -85,6 +85,30 @@ model_list: output_cost_per_token: 0.00000044 cache_read_input_token_cost: 0.000000022 + # --- PublicAI (publicly developed / sovereign models) --- + # OpenAI-compatible gateway for public open models. Apertus is the Swiss AI + # Initiative's fully-open model (Apache-2.0: weights, code, and training data). + # Pricing from PublicAI /v1/models (per 1M): 8b $0.10/$0.20, 70b $0.82/$2.92. + # Member-facing name drops the upstream `swiss-ai/` owner prefix for brevity; + # the litellm_params.model keeps the full PublicAI id. + - model_name: publicai/apertus-v1.5-8b + litellm_params: + model: openai/swiss-ai/apertus-v1.5-8b + api_base: https://api.publicai.co/v1 + api_key: os.environ/PUBLICAI_API_KEY + model_info: + input_cost_per_token: 0.00000010 + output_cost_per_token: 0.00000020 + + - model_name: publicai/apertus-v1.5-70b + litellm_params: + model: openai/swiss-ai/apertus-v1.5-70b + api_base: https://api.publicai.co/v1 + api_key: os.environ/PUBLICAI_API_KEY + model_info: + input_cost_per_token: 0.00000082 + output_cost_per_token: 0.00000292 + # Fallback: if DeepSeek is down, use GPT-OSS # NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run # longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek