diff --git a/config.yaml b/config.yaml index d88c59a..56d3104 100644 --- a/config.yaml +++ b/config.yaml @@ -1,26 +1,22 @@ # LiteLLM config for inference.coop -# MVP config: Tinfoil only (TEE-protected, privacy-first) -# Two models: DeepSeek V4 Flash (default, best agents) + GPT-OSS 120B (cheapest) -# Provider key set via environment variable or Admin UI +# TEST config: Ollama Cloud (existing subscription) — plumbing smoke test +# Uses Ollama Cloud's OpenAI-compatible endpoint (/v1) via LiteLLM's openai/ provider. +# Provider key set via environment variable OLLAMA_API_KEY. model_list: # DeepSeek V4 Flash — default model - # Best agent performance (Terminal Bench 2.1: 82.7, DeepSWE: 54.4) - # MoE, 1M context, tool calling, $0.30/$0.70 per M tokens - model_name: deepseek-v4-flash litellm_params: - model: openai/deepseek-v4-flash - api_base: https://api.tinfoil.sh/v1 - api_key: os.environ/TINFOIL_API_KEY + model: openai/deepseek-v4-flash:0731 + api_base: https://ollama.com/v1 + api_key: os.environ/OLLAMA_API_KEY # GPT-OSS 120B — lightweight fallback - # Built for agentic workflows, web search + code execution - # Apache 2.0, $0.15/$0.60 per M tokens - model_name: gpt-oss-120b litellm_params: - model: openai/gpt-oss-120b - api_base: https://api.tinfoil.sh/v1 - api_key: os.environ/TINFOIL_API_KEY + model: openai/gpt-oss:120b + api_base: https://ollama.com/v1 + api_key: os.environ/OLLAMA_API_KEY # Fallback: if DeepSeek is down, use GPT-OSS router_settings: @@ -39,4 +35,4 @@ litellm_settings: salt_key: os.environ/LITELLM_SALT_KEY drop_params: true num_threads: 4 - request_timeout: 30 \ No newline at end of file + request_timeout: 30