Switch to Ollama Cloud (OpenAI-compatible /v1 endpoint) for plumbing smoke test
This commit is contained in:
1 parent
41068a6264
commit
2ce5c66b9b
1 file changed
+10
-14
+10
-14
@@ -1,26 +1,22 @@
|
||||
# LiteLLM config for inference.coop
|
||||
# MVP config: Tinfoil only (TEE-protected, privacy-first)
|
||||
# Two models: DeepSeek V4 Flash (default, best agents) + GPT-OSS 120B (cheapest)
|
||||
# Provider key set via environment variable or Admin UI
|
||||
# TEST config: Ollama Cloud (existing subscription) — plumbing smoke test
|
||||
# Uses Ollama Cloud's OpenAI-compatible endpoint (/v1) via LiteLLM's openai/ provider.
|
||||
# Provider key set via environment variable OLLAMA_API_KEY.
|
||||
|
||||
model_list:
|
||||
# DeepSeek V4 Flash — default model
|
||||
# Best agent performance (Terminal Bench 2.1: 82.7, DeepSWE: 54.4)
|
||||
# MoE, 1M context, tool calling, $0.30/$0.70 per M tokens
|
||||
- model_name: deepseek-v4-flash
|
||||
litellm_params:
|
||||
model: openai/deepseek-v4-flash
|
||||
api_base: https://api.tinfoil.sh/v1
|
||||
api_key: os.environ/TINFOIL_API_KEY
|
||||
model: openai/deepseek-v4-flash:0731
|
||||
api_base: https://ollama.com/v1
|
||||
api_key: os.environ/OLLAMA_API_KEY
|
||||
|
||||
# GPT-OSS 120B — lightweight fallback
|
||||
# Built for agentic workflows, web search + code execution
|
||||
# Apache 2.0, $0.15/$0.60 per M tokens
|
||||
- model_name: gpt-oss-120b
|
||||
litellm_params:
|
||||
model: openai/gpt-oss-120b
|
||||
api_base: https://api.tinfoil.sh/v1
|
||||
api_key: os.environ/TINFOIL_API_KEY
|
||||
model: openai/gpt-oss:120b
|
||||
api_base: https://ollama.com/v1
|
||||
api_key: os.environ/OLLAMA_API_KEY
|
||||
|
||||
# Fallback: if DeepSeek is down, use GPT-OSS
|
||||
router_settings:
|
||||
@@ -39,4 +35,4 @@ litellm_settings:
|
||||
salt_key: os.environ/LITELLM_SALT_KEY
|
||||
drop_params: true
|
||||
num_threads: 4
|
||||
request_timeout: 30
|
||||
request_timeout: 30
|
||||
Reference in new issue
Block a user