186 lines
7.4 KiB
YAML
186 lines
7.4 KiB
YAML
# LiteLLM config for inference.coop
|
|
# PRODUCTION config: Tinfoil (TEE-protected inference)
|
|
# LiteLLM talks plaintext OpenAI to a local Tinfoil proxy (127.0.0.1:3301),
|
|
# which verifies the enclave attestation and encrypts request/response bodies
|
|
# with EHBP (HPKE) before forwarding to the Tinfoil enclave. Provider key set
|
|
# via environment variable TINFOIL_API_KEY.
|
|
#
|
|
# Pricing (model_info) is per-token, derived from Tinfoil's per-1M rates:
|
|
# tinfoil/deepseek-v4-1-flash: $0.65 in / $1.45 out / $0.13 cached per 1M
|
|
# tinfoil/gpt-oss-120b: $0.15 in / $0.60 out per 1M
|
|
# tinfoil/glm-5-3-flash: $0.40 in / $1.25 out / $0.10 cached per 1M
|
|
# Without these, LiteLLM can't price these custom models, so spend shows $0.00
|
|
# and the team budget cap is not enforced.
|
|
#
|
|
# SYNC: the per-model pricing here must mirror the user-facing model list in
|
|
# co-op/docs/models.md (and vice-versa). When a rate changes, update BOTH this
|
|
# file and models.md — this config is the authoritative source of the numbers;
|
|
# models.md is the member-facing mirror.
|
|
|
|
model_list:
|
|
# DeepSeek V4.1 Flash — default model (replaces V4 Flash, deprecated 2026-09-15)
|
|
- model_name: tinfoil/deepseek-v4-1-flash
|
|
litellm_params:
|
|
model: openai/deepseek-v4-1-flash
|
|
api_base: http://127.0.0.1:3301/v1
|
|
api_key: os.environ/TINFOIL_API_KEY
|
|
model_info:
|
|
input_cost_per_token: 0.00000065
|
|
output_cost_per_token: 0.00000145
|
|
cache_read_input_token_cost: 0.00000013
|
|
|
|
# GPT-OSS 120B — lightweight fallback
|
|
- model_name: tinfoil/gpt-oss-120b
|
|
litellm_params:
|
|
model: openai/gpt-oss-120b
|
|
api_base: http://127.0.0.1:3301/v1
|
|
api_key: os.environ/TINFOIL_API_KEY
|
|
model_info:
|
|
input_cost_per_token: 0.00000015
|
|
output_cost_per_token: 0.00000060
|
|
|
|
# GLM-5.3 Flash — fast, efficient MoE model
|
|
- model_name: tinfoil/glm-5-3-flash
|
|
litellm_params:
|
|
model: openai/glm-5-3-flash
|
|
api_base: http://127.0.0.1:3301/v1
|
|
api_key: os.environ/TINFOIL_API_KEY
|
|
model_info:
|
|
input_cost_per_token: 0.00000040
|
|
output_cost_per_token: 0.00000125
|
|
cache_read_input_token_cost: 0.00000010
|
|
|
|
# --- GreenPT (renewable energy) ---
|
|
# OpenAI-compatible, EU-hosted on 100% renewable energy. No TEE proxy needed —
|
|
# LiteLLM talks to GreenPT directly. Pricing from GreenPT /v1/pricing endpoint.
|
|
# green-r: $0.35 in / $0.95 out per 1M (128k ctx)
|
|
# green-l: $0.25 in / $0.80 out per 1M (128k ctx)
|
|
- model_name: greenpt/green-r
|
|
litellm_params:
|
|
model: openai/green-r
|
|
api_base: https://api.greenpt.ai/v1
|
|
api_key: os.environ/GREENPT_API_KEY
|
|
model_info:
|
|
input_cost_per_token: 0.00000035
|
|
output_cost_per_token: 0.00000095
|
|
|
|
- model_name: greenpt/green-l
|
|
litellm_params:
|
|
model: openai/green-l
|
|
api_base: https://api.greenpt.ai/v1
|
|
api_key: os.environ/GREENPT_API_KEY
|
|
model_info:
|
|
input_cost_per_token: 0.00000025
|
|
output_cost_per_token: 0.00000080
|
|
|
|
# GLM-5.3 Flash on GreenPT (note the dotted slug — distinct from Tinfoil's
|
|
# glm-5-3-flash). $0.11 in / $0.44 out / $0.022 cached per 1M.
|
|
- model_name: greenpt/glm-5.3-flash
|
|
litellm_params:
|
|
model: openai/glm-5.3-flash
|
|
api_base: https://api.greenpt.ai/v1
|
|
api_key: os.environ/GREENPT_API_KEY
|
|
model_info:
|
|
input_cost_per_token: 0.00000011
|
|
output_cost_per_token: 0.00000044
|
|
cache_read_input_token_cost: 0.000000022
|
|
|
|
# GLM-5.3 (full flagship) on GreenPT — strongest agentic model in the green
|
|
# tier. $1.10 in / $4.40 out per 1M, 1M context.
|
|
- model_name: greenpt/glm-5.3
|
|
litellm_params:
|
|
model: openai/glm-5.3
|
|
api_base: https://api.greenpt.ai/v1
|
|
api_key: os.environ/GREENPT_API_KEY
|
|
model_info:
|
|
input_cost_per_token: 0.0000011
|
|
output_cost_per_token: 0.0000044
|
|
|
|
# --- Audio models (Tinfoil, inside the TEE) ---
|
|
# NOT chat models: these power /v1/audio/* routes (STT for dictation, TTS for
|
|
# voice mode). They stay VISIBLE in the API model list so developer members
|
|
# can call them, but the portal filters them out of the chat-facing list
|
|
# (Open WebUI) by their `mode` field, so they never appear as pickable chat
|
|
# options. Pricing per Tinfoil catalog: whisper $0.05/1M in; voxtral-tts
|
|
# listed $0/$0 (verify on invoice — may be beta-free).
|
|
|
|
|
|
# --- PublicAI (publicly developed / sovereign models) ---
|
|
# OpenAI-compatible gateway for public open models. Apertus is the Swiss AI
|
|
# Initiative's fully-open model (Apache-2.0: weights, code, and training data).
|
|
# Pricing from PublicAI /v1/models (per 1M): 8b $0.10/$0.20, 70b $0.82/$2.92.
|
|
# Member-facing name drops the upstream `swiss-ai/` owner prefix for brevity;
|
|
# the litellm_params.model keeps the full PublicAI id.
|
|
- model_name: publicai/apertus-v1.5-8b
|
|
litellm_params:
|
|
model: openai/swiss-ai/apertus-v1.5-8b
|
|
api_base: https://api.publicai.co/v1
|
|
api_key: os.environ/PUBLICAI_API_KEY
|
|
model_info:
|
|
input_cost_per_token: 0.00000010
|
|
output_cost_per_token: 0.00000020
|
|
|
|
- model_name: publicai/apertus-v1.5-70b
|
|
litellm_params:
|
|
model: openai/swiss-ai/apertus-v1.5-70b
|
|
api_base: https://api.publicai.co/v1
|
|
api_key: os.environ/PUBLICAI_API_KEY
|
|
model_info:
|
|
input_cost_per_token: 0.00000082
|
|
output_cost_per_token: 0.00000292
|
|
|
|
# --- Audio models (Tinfoil, in-TEE) ---
|
|
# Used for chat dictation (STT) and voice mode (TTS). These are NOT chat
|
|
# models: the member portal's /v1/models filters them out of the chat model
|
|
# list (see AUDIO_MODEL_PREFIXES there) so members don't see them in the
|
|
# picker, but they remain fully callable via /v1/audio/* for API users.
|
|
# whisper-large-v3-turbo: $0.05/1M input tokens. voxtral-tts: listed $0/$0.
|
|
# NOTE: LiteLLM routes audio by the openai/ provider + the endpoint called
|
|
# (/v1/audio/transcriptions or /v1/audio/speech) — there is no
|
|
# "audio_transcription/" provider prefix.
|
|
- model_name: tinfoil/whisper-large-v3-turbo
|
|
litellm_params:
|
|
model: openai/whisper-large-v3-turbo
|
|
api_base: http://127.0.0.1:3301/v1
|
|
api_key: os.environ/TINFOIL_API_KEY
|
|
model_info:
|
|
mode: audio_transcription
|
|
input_cost_per_token: 0.00000005
|
|
|
|
- model_name: tinfoil/voxtral-tts
|
|
litellm_params:
|
|
model: openai/voxtral-tts
|
|
api_base: http://127.0.0.1:3301/v1
|
|
api_key: os.environ/TINFOIL_API_KEY
|
|
model_info:
|
|
mode: audio_speech
|
|
|
|
# Fallback: if DeepSeek is down, use GPT-OSS
|
|
# NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run
|
|
# longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek
|
|
# requests (falling back to GPT-OSS), while Tinfoil still completed and billed
|
|
# the abandoned generation. Result: Tinfoil billed ~2x the tokens LiteLLM logged.
|
|
# Fix: a generous timeout (180s) so generations complete inside LiteLLM and get
|
|
# logged; and a retry_policy that retries transient connection errors (base
|
|
# num_retries) but NEVER retries timeouts (TimeoutErrorRetries: 0) — a timed-out
|
|
# generation is already billed by Tinfoil, so re-hitting it re-bills.
|
|
router_settings:
|
|
num_retries: 2
|
|
timeout: 180
|
|
retry_policy:
|
|
TimeoutErrorRetries: 0
|
|
fallbacks:
|
|
- tinfoil/deepseek-v4-1-flash:
|
|
- tinfoil/gpt-oss-120b
|
|
|
|
general_settings:
|
|
master_key: os.environ/LITELLM_MASTER_KEY
|
|
database_url: os.environ/DATABASE_URL
|
|
store_model_in_db: true
|
|
|
|
litellm_settings:
|
|
salt_key: os.environ/LITELLM_SALT_KEY
|
|
drop_params: true
|
|
num_threads: 4
|
|
request_timeout: 180
|