Add Tinfoil audio models (whisper STT, voxtral-tts) for voice mode + API use
This commit is contained in:
1 parent
5ea2256776
commit
8f42be988e
1 file changed
+52
+52
@@ -96,6 +96,32 @@ model_list:
|
|||||||
input_cost_per_token: 0.0000011
|
input_cost_per_token: 0.0000011
|
||||||
output_cost_per_token: 0.0000044
|
output_cost_per_token: 0.0000044
|
||||||
|
|
||||||
|
# --- Audio models (Tinfoil, inside the TEE) ---
|
||||||
|
# NOT chat models: these power /v1/audio/* routes (STT for dictation, TTS for
|
||||||
|
# voice mode). They stay VISIBLE in the API model list so developer members
|
||||||
|
# can call them, but the portal filters them out of the chat-facing list
|
||||||
|
# (Open WebUI) by their `mode` field, so they never appear as pickable chat
|
||||||
|
# options. Pricing per Tinfoil catalog: whisper $0.05/1M in; voxtral-tts
|
||||||
|
# listed $0/$0 (verify on invoice — may be beta-free).
|
||||||
|
- model_name: tinfoil/whisper-large-v3-turbo
|
||||||
|
litellm_params:
|
||||||
|
model: audio_transcription/whisper-large-v3-turbo
|
||||||
|
api_base: http://127.0.0.1:3301/v1
|
||||||
|
api_key: os.environ/TINFOIL_API_KEY
|
||||||
|
model_info:
|
||||||
|
mode: audio_transcription
|
||||||
|
input_cost_per_token: 0.00000005
|
||||||
|
|
||||||
|
- model_name: tinfoil/voxtral-tts
|
||||||
|
litellm_params:
|
||||||
|
model: audio_speech/voxtral-tts
|
||||||
|
api_base: http://127.0.0.1:3301/v1
|
||||||
|
api_key: os.environ/TINFOIL_API_KEY
|
||||||
|
model_info:
|
||||||
|
mode: audio_speech
|
||||||
|
input_cost_per_token: 0.0
|
||||||
|
output_cost_per_token: 0.0
|
||||||
|
|
||||||
# --- PublicAI (publicly developed / sovereign models) ---
|
# --- PublicAI (publicly developed / sovereign models) ---
|
||||||
# OpenAI-compatible gateway for public open models. Apertus is the Swiss AI
|
# OpenAI-compatible gateway for public open models. Apertus is the Swiss AI
|
||||||
# Initiative's fully-open model (Apache-2.0: weights, code, and training data).
|
# Initiative's fully-open model (Apache-2.0: weights, code, and training data).
|
||||||
@@ -120,6 +146,32 @@ model_list:
|
|||||||
input_cost_per_token: 0.00000082
|
input_cost_per_token: 0.00000082
|
||||||
output_cost_per_token: 0.00000292
|
output_cost_per_token: 0.00000292
|
||||||
|
|
||||||
|
# --- Audio models (Tinfoil, in-TEE) ---
|
||||||
|
# Used for chat dictation (STT) and voice mode (TTS). These are NOT chat
|
||||||
|
# models: the member portal's /v1/models filters them out of the chat model
|
||||||
|
# list (see AUDIO_MODEL_PREFIXES there) so members don't see them in the
|
||||||
|
# picker, but they remain fully callable via /v1/audio/* for API users.
|
||||||
|
# whisper-large-v3-turbo: $0.05/1M input tokens. voxtral-tts: listed $0/$0.
|
||||||
|
# NOTE: LiteLLM routes audio by the openai/ provider + the endpoint called
|
||||||
|
# (/v1/audio/transcriptions or /v1/audio/speech) — there is no
|
||||||
|
# "audio_transcription/" provider prefix.
|
||||||
|
- model_name: tinfoil/whisper-large-v3-turbo
|
||||||
|
litellm_params:
|
||||||
|
model: openai/whisper-large-v3-turbo
|
||||||
|
api_base: http://127.0.0.1:3301/v1
|
||||||
|
api_key: os.environ/TINFOIL_API_KEY
|
||||||
|
model_info:
|
||||||
|
mode: audio_transcription
|
||||||
|
input_cost_per_token: 0.00000005
|
||||||
|
|
||||||
|
- model_name: tinfoil/voxtral-tts
|
||||||
|
litellm_params:
|
||||||
|
model: openai/voxtral-tts
|
||||||
|
api_base: http://127.0.0.1:3301/v1
|
||||||
|
api_key: os.environ/TINFOIL_API_KEY
|
||||||
|
model_info:
|
||||||
|
mode: audio_speech
|
||||||
|
|
||||||
# Fallback: if DeepSeek is down, use GPT-OSS
|
# Fallback: if DeepSeek is down, use GPT-OSS
|
||||||
# NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run
|
# NOTE: DeepSeek V4.1 Flash is a reasoning model — generations routinely run
|
||||||
# longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek
|
# longer than 30s. A 30s timeout caused LiteLLM to abandon in-flight DeepSeek
|
||||||
|
|||||||
Reference in new issue
Block a user