Initial commit: LiteLLM Cloudron app package for Inference Cooperative

This commit is contained in:
inference-bot committed 2026-09-03 21:55:54 -06:00
commit b9b4620e52
7 files changed
+375

No files matched your search

+42
View File
@@ -0,0 +1,42 @@
# LiteLLM config for inference.coop
# MVP config: Tinfoil only (TEE-protected, privacy-first)
# Two models: DeepSeek V4 Flash (default, best agents) + GPT-OSS 120B (cheapest)
# Provider key set via environment variable or Admin UI
model_list:
# DeepSeek V4 Flash — default model
# Best agent performance (Terminal Bench 2.1: 82.7, DeepSWE: 54.4)
# MoE, 1M context, tool calling, $0.30/$0.70 per M tokens
- model_name: deepseek-v4-flash
litellm_params:
model: openai/deepseek-v4-flash
api_base: https://api.tinfoil.sh/v1
api_key: os.environ/TINFOIL_API_KEY
# GPT-OSS 120B — lightweight fallback
# Built for agentic workflows, web search + code execution
# Apache 2.0, $0.15/$0.60 per M tokens
- model_name: gpt-oss-120b
litellm_params:
model: openai/gpt-oss-120b
api_base: https://api.tinfoil.sh/v1
api_key: os.environ/TINFOIL_API_KEY
# Fallback: if DeepSeek is down, use GPT-OSS
router_settings:
num_retries: 2
timeout: 30
fallbacks:
- deepseek-v4-flash:
- gpt-oss-120b
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
database_url: os.environ/DATABASE_URL
store_model_in_db: true
litellm_settings:
salt_key: os.environ/LITELLM_SALT_KEY
drop_params: true
num_threads: 4
request_timeout: 30