diff --git a/config.yaml b/config.yaml index d1f04c6..b391962 100644 --- a/config.yaml +++ b/config.yaml @@ -96,6 +96,32 @@ model_list: input_cost_per_token: 0.0000011 output_cost_per_token: 0.0000044 + # --- Audio models (Tinfoil, inside the TEE) --- + # NOT chat models: these power /v1/audio/* routes (STT for dictation, TTS for + # voice mode). They stay VISIBLE in the API model list so developer members + # can call them, but the portal filters them out of the chat-facing list + # (Open WebUI) by their `mode` field, so they never appear as pickable chat + # options. Pricing per Tinfoil catalog: whisper $0.05/1M in; voxtral-tts + # listed $0/$0 (verify on invoice — may be beta-free). + - model_name: tinfoil/whisper-large-v3-turbo + litellm_params: + model: audio_transcription/whisper-large-v3-turbo + api_base: http://127.0.0.1:3301/v1 + api_key: os.environ/TINFOIL_API_KEY + model_info: + mode: audio_transcription + input_cost_per_token: 0.00000005 + + - model_name: tinfoil/voxtral-tts + litellm_params: + model: audio_speech/voxtral-tts + api_base: http://127.0.0.1:3301/v1 + api_key: os.environ/TINFOIL_API_KEY + model_info: + mode: audio_speech + input_cost_per_token: 0.0 + output_cost_per_token: 0.0 + # --- PublicAI (publicly developed / sovereign models) --- # OpenAI-compatible gateway for public open models. Apertus is the Swiss AI # Initiative's fully-open model (Apache-2.0: weights, code, and training data). @@ -120,6 +146,32 @@ model_list: input_cost_per_token: 0.00000082 output_cost_per_token: 0.00000292 + # --- Audio models (Tinfoil, in-TEE) --- + # Used for chat dictation (STT) and voice mode (TTS). These are NOT chat + # models: the member portal's /v1/models filters them out of the chat model + # list (see AUDIO_MODEL_PREFIXES there) so members don't see them in the + # picker, but they remain fully callable via /v1/audio/* for API users. + # whisper-large-v3-turbo: $0.05/1M input tokens. voxtral-tts: listed $0/$0. + # NOTE: LiteLLM routes audio by the openai/ provider + the endpoint called + # (/v1/audio/transcriptions or /v1/audio/speech) — there is no + # "audio_transcription/" provider prefix. + - model_name: tinfoil/whisper-large-v3-turbo + litellm_params: + model: openai/whisper-large-v3-turbo + api_base: http://127.0.0.1:3301/v1 + api_key: os.environ/TINFOIL_API_KEY + model_info: + mode: audio_transcription + input_cost_per_token: 0.00000005 + + - model_name: tinfoil/voxtral-tts + litellm_params: + model: openai/voxtral-tts + api_base: http://127.0.0.1:3301/v1 + api_key: os.environ/TINFOIL_API_KEY + model_info: + mode: audio_speech + # Fallback: if DeepSeek is down, use GPT-OSS # NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run # longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek