Add Tinfoil audio models (whisper STT, voxtral-tts) for voice mode + API use

This commit is contained in:
inference-bot committed 2026-09-27 11:10:51 -06:00
1 parent 5ea2256776
commit 8f42be988e
1 file changed
+52
+52
View File
@@ -96,6 +96,32 @@ model_list:
input_cost_per_token: 0.0000011 input_cost_per_token: 0.0000011
output_cost_per_token: 0.0000044 output_cost_per_token: 0.0000044
# --- Audio models (Tinfoil, inside the TEE) ---
# NOT chat models: these power /v1/audio/* routes (STT for dictation, TTS for
# voice mode). They stay VISIBLE in the API model list so developer members
# can call them, but the portal filters them out of the chat-facing list
# (Open WebUI) by their `mode` field, so they never appear as pickable chat
# options. Pricing per Tinfoil catalog: whisper $0.05/1M in; voxtral-tts
# listed $0/$0 (verify on invoice — may be beta-free).
- model_name: tinfoil/whisper-large-v3-turbo
litellm_params:
model: audio_transcription/whisper-large-v3-turbo
api_base: http://127.0.0.1:3301/v1
api_key: os.environ/TINFOIL_API_KEY
model_info:
mode: audio_transcription
input_cost_per_token: 0.00000005
- model_name: tinfoil/voxtral-tts
litellm_params:
model: audio_speech/voxtral-tts
api_base: http://127.0.0.1:3301/v1
api_key: os.environ/TINFOIL_API_KEY
model_info:
mode: audio_speech
input_cost_per_token: 0.0
output_cost_per_token: 0.0
# --- PublicAI (publicly developed / sovereign models) --- # --- PublicAI (publicly developed / sovereign models) ---
# OpenAI-compatible gateway for public open models. Apertus is the Swiss AI # OpenAI-compatible gateway for public open models. Apertus is the Swiss AI
# Initiative's fully-open model (Apache-2.0: weights, code, and training data). # Initiative's fully-open model (Apache-2.0: weights, code, and training data).
@@ -120,6 +146,32 @@ model_list:
input_cost_per_token: 0.00000082 input_cost_per_token: 0.00000082
output_cost_per_token: 0.00000292 output_cost_per_token: 0.00000292
# --- Audio models (Tinfoil, in-TEE) ---
# Used for chat dictation (STT) and voice mode (TTS). These are NOT chat
# models: the member portal's /v1/models filters them out of the chat model
# list (see AUDIO_MODEL_PREFIXES there) so members don't see them in the
# picker, but they remain fully callable via /v1/audio/* for API users.
# whisper-large-v3-turbo: $0.05/1M input tokens. voxtral-tts: listed $0/$0.
# NOTE: LiteLLM routes audio by the openai/ provider + the endpoint called
# (/v1/audio/transcriptions or /v1/audio/speech) — there is no
# "audio_transcription/" provider prefix.
- model_name: tinfoil/whisper-large-v3-turbo
litellm_params:
model: openai/whisper-large-v3-turbo
api_base: http://127.0.0.1:3301/v1
api_key: os.environ/TINFOIL_API_KEY
model_info:
mode: audio_transcription
input_cost_per_token: 0.00000005
- model_name: tinfoil/voxtral-tts
litellm_params:
model: openai/voxtral-tts
api_base: http://127.0.0.1:3301/v1
api_key: os.environ/TINFOIL_API_KEY
model_info:
mode: audio_speech
# Fallback: if DeepSeek is down, use GPT-OSS # Fallback: if DeepSeek is down, use GPT-OSS
# NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run # NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run
# longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek # longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek