From b9b4620e52c5f383ee04de40ed4eaef3894ee206 Mon Sep 17 00:00:00 2001 From: inference-bot Date: Thu, 3 Sep 2026 21:55:54 -0600 Subject: [PATCH] Initial commit: LiteLLM Cloudron app package for Inference Cooperative --- .dockerignore | 5 ++ CloudronManifest.json | 23 +++++++ Dockerfile | 25 +++++++ README.md | 151 ++++++++++++++++++++++++++++++++++++++++++ config.yaml | 42 ++++++++++++ logo.png | Bin 0 -> 6756 bytes start.sh | 129 ++++++++++++++++++++++++++++++++++++ 7 files changed, 375 insertions(+) create mode 100644 .dockerignore create mode 100644 CloudronManifest.json create mode 100644 Dockerfile create mode 100644 README.md create mode 100644 config.yaml create mode 100644 logo.png create mode 100644 start.sh diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..43620d2 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,5 @@ +node_modules +.git +*.md +Dockerfile +.dockerignore \ No newline at end of file diff --git a/CloudronManifest.json b/CloudronManifest.json new file mode 100644 index 0000000..be27812 --- /dev/null +++ b/CloudronManifest.json @@ -0,0 +1,23 @@ +{ + "id": "ai.coop.litellm", + "title": "LiteLLM Gateway", + "author": "inference.coop", + "description": "LiteLLM AI Gateway — OpenAI-compatible proxy with per-user API keys, usage tracking, rate limiting, and model routing. Packaged for Cloudron with SSO/OIDC and PostgreSQL.", + "tagline": "AI gateway for cooperative inference", + "version": "1.0.0", + "upstreamVersion": "1.74.0", + "healthCheckPath": "/health/readiness", + "httpPort": 4000, + "manifestVersion": 2, + "website": "https://litellm.ai", + "contactEmail": "botbot@hermes", + "icon": "file://logo.png", + "addons": { + "postgresql": {}, + "redis": {}, + "localstorage": {} + }, + "tags": ["ai", "gateway", "proxy", "api"], + "mediaLinks": [], + "changelog": "Initial package for inference.coop" +} \ No newline at end of file diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..fd4389b --- /dev/null +++ b/Dockerfile @@ -0,0 +1,25 @@ +FROM ghcr.io/berriai/litellm:main-v1.74.0-stable + +# Cloudron runs apps as the 'cloudron' user (uid 1000) by default. +# LiteLLM's official image runs as root; we switch to cloudron for security. +# However, LiteLLM needs to write to /app/data for its database/config. + +USER root + +# Install supervisor to manage processes (LiteLLM + optional cron) +RUN pip install --no-cache-dir supervisor + +# Create the data directory and set ownership +RUN mkdir -p /app/data && chown -R cloudron:cloudron /app/data + +# Create a startup script that configures LiteLLM with Cloudron addons +COPY start.sh /app/code/start.sh +RUN chmod +x /app/code/start.sh && chown cloudron:cloudron /app/code/start.sh + +# Default config — will be overridden by Cloudron env vars at runtime +COPY config.yaml /app/data/config.yaml +RUN chown cloudron:cloudron /app/data/config.yaml + +USER cloudron + +CMD ["/app/code/start.sh"] \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..e96d558 --- /dev/null +++ b/README.md @@ -0,0 +1,151 @@ +# LiteLLM Gateway — Cloudron App + +A Cloudron app package for [LiteLLM](https://litellm.ai/), the open-source AI gateway, configured for the **Inference Cooperative**. + +LiteLLM is the single gateway between members and the inference backends. It handles per-member API keys, usage metering, rate limiting (via budget caps), and model routing — all behind one OpenAI-compatible endpoint. + +## What this provides + +- **LiteLLM proxy** with an OpenAI-compatible API on port 4000 +- **Per-member virtual API keys** with budget caps and rate limits +- **Usage / spend tracking** per key, per model +- **Model routing** to Tinfoil (TEE-protected inference) +- **Cloudron PostgreSQL** for keys, teams, and spend logs +- **Cloudron Redis** for rate limiting and caching +- **Automatic SSL, backups, and sandboxing** (Cloudron-managed) +- **Master key + salt key** auto-generated on first start, persisted in `/app/data` + +## Architecture + +``` +Members → Open WebUI / LobeChat → LiteLLM (this app) → Tinfoil (TEE) +``` + +LiteLLM is the control plane (keys, metering, routing). It does **not** run models itself — it forwards to Tinfoil, which runs open models inside trusted execution environments (TEEs). + +## Files + +- `CloudronManifest.json` — Cloudron app manifest (addons, ports, metadata) +- `Dockerfile` — wraps LiteLLM's official image for Cloudron's environment +- `start.sh` — startup script that wires up Cloudron's Postgres/Redis and writes the default config +- `config.yaml` — default LiteLLM config (Tinfoil-only, two models) +- `logo.png` — app icon + +## Models (MVP) + +Two models, both TEE-protected via Tinfoil: + +| Model | Role | Price (in/out per M tokens) | +|-------|------|------------------------------| +| `deepseek-v4-flash` | Default — best agent performance, 1M context | $0.30 / $0.70 | +| `gpt-oss-120b` | Fallback — cheapest, built for agentic workflows | $0.15 / $0.60 | + +If DeepSeek is down, LiteLLM automatically falls back to GPT-OSS. + +--- + +## Building and installing + +This package uses Cloudron's **on-server build** (no local Docker, no registry). The Cloudron CLI uploads the source and Cloudron builds the image on the server. + +### Prerequisites + +- The [Cloudron CLI](https://docs.cloudron.io/packaging/cli/) installed on a machine with access to the Cloudron server +- A Cloudron instance with the `inference.coop` domain + +### Install + +```bash +# 1. Install the Cloudron CLI (once) +npm install -g cloudron-cli + +# 2. Log in to the Cloudron +cloudron login my.inference.coop + +# 3. From this directory, install (builds on the server) +cloudron install --location gateway +``` + +`--location gateway` installs the app at `gateway.inference.coop`. Omit `--location` to be prompted. + +### Configure + +After install, set the Tinfoil API key in Cloudron → app → Settings → Environment Variables: + +- `TINFOIL_API_KEY` — your Tinfoil API key (from https://tinfoil.sh/) + +The following are auto-generated and should **not** be set manually: + +- `LITELLM_MASTER_KEY` — generated on first start, stored in `/app/data/.master_key` +- `LITELLM_SALT_KEY` — generated on first start, stored in `/app/data/.salt_key` +- `DATABASE_URL` — from Cloudron's PostgreSQL addon +- `REDIS_HOST`, `REDIS_PORT`, `REDIS_PASSWORD` — from Cloudron's Redis addon + +### Access the admin UI + +1. Open `https://gateway.inference.coop/ui` +2. Log in with username `admin` and the master key as password +3. Find the master key: `cloudron exec --app cat /app/data/.master_key` + +--- + +## Updating LiteLLM + +This is the routine maintenance path. LiteLLM releases frequently; updating is a two-step change. + +### 1. Bump the base image tag + +In `Dockerfile`, change the pinned version: + +```dockerfile +FROM ghcr.io/berriai/litellm:main-v1.74.0-stable +# ^^^^^^ bump this +``` + +### 2. Bump the manifest version + +In `CloudronManifest.json`, update: + +```json +"version": "1.0.0", +"upstreamVersion": "1.74.0" +``` + +### 3. Redeploy + +```bash +cloudron update --app +``` + +Cloudron rebuilds from source and applies the update. Data (keys, spend logs) persists in PostgreSQL. + +### Before updating + +- Check the [LiteLLM changelog](https://github.com/BerriAI/litellm/releases) for breaking changes +- The `start.sh` and `config.yaml` may need adjustment if LiteLLM changes its env-var or config schema +- Test on a staging instance if the version jump is large + +--- + +## Environment variables + +| Variable | Required | Purpose | +|----------|----------|---------| +| `TINFOIL_API_KEY` | Yes | Tinfoil API key for TEE inference | + +## Connecting Open WebUI + +In Open WebUI settings: + +- **API Base URL:** `https://gateway.inference.coop/v1` +- **API Key:** a virtual key created in LiteLLM's Admin UI + +## OIDC / SSO (future) + +The current package uses LiteLLM's built-in admin/master-key auth, which is sufficient for the MVP. Full OIDC SSO (admin UI behind Cloudron login) is a TODO — add the `"oidc": {}` addon to the manifest and configure LiteLLM's SSO hooks per https://docs.litellm.ai/docs/proxy/custom_sso. + +## Related repos + +- `co-op/org-dev` — the full inference.coop design doc and deploy scripts +- `co-op/website` — the public landing page +- `co-op/design-assets` — logos, fonts, brand materials diff --git a/config.yaml b/config.yaml new file mode 100644 index 0000000..d88c59a --- /dev/null +++ b/config.yaml @@ -0,0 +1,42 @@ +# LiteLLM config for inference.coop +# MVP config: Tinfoil only (TEE-protected, privacy-first) +# Two models: DeepSeek V4 Flash (default, best agents) + GPT-OSS 120B (cheapest) +# Provider key set via environment variable or Admin UI + +model_list: + # DeepSeek V4 Flash — default model + # Best agent performance (Terminal Bench 2.1: 82.7, DeepSWE: 54.4) + # MoE, 1M context, tool calling, $0.30/$0.70 per M tokens + - model_name: deepseek-v4-flash + litellm_params: + model: openai/deepseek-v4-flash + api_base: https://api.tinfoil.sh/v1 + api_key: os.environ/TINFOIL_API_KEY + + # GPT-OSS 120B — lightweight fallback + # Built for agentic workflows, web search + code execution + # Apache 2.0, $0.15/$0.60 per M tokens + - model_name: gpt-oss-120b + litellm_params: + model: openai/gpt-oss-120b + api_base: https://api.tinfoil.sh/v1 + api_key: os.environ/TINFOIL_API_KEY + +# Fallback: if DeepSeek is down, use GPT-OSS +router_settings: + num_retries: 2 + timeout: 30 + fallbacks: + - deepseek-v4-flash: + - gpt-oss-120b + +general_settings: + master_key: os.environ/LITELLM_MASTER_KEY + database_url: os.environ/DATABASE_URL + store_model_in_db: true + +litellm_settings: + salt_key: os.environ/LITELLM_SALT_KEY + drop_params: true + num_threads: 4 + request_timeout: 30 \ No newline at end of file diff --git a/logo.png b/logo.png new file mode 100644 index 0000000000000000000000000000000000000000..63278129959f29921b57decddc30575f504e3550 GIT binary patch literal 6756 zcmb7Jc|4SB-~ZkB3^9_WvE>|V3MERkGfyEI9b>B#i8fnAr4li>6WNaz*|U|1G>E7y zlbmu|RhCj&+K9?jgNZToUbmjl^PJ~>-uHPwuYcxu-E+-#U%&PH{r$wVepxR|n?nNt z*$uW- zwXdVmUlL&uDrKKwm=LYPzcKP>^Qp@rj~-e0aFhDt@8$-dj5G5O&DWM|i``$5GJk~Dtf6IIzM*bl9tnzcU_oo3P4wWiHF&pYOewtZ6wm7<3^YxL*cdwp{#h<3SY)*rF z$3$zN;W9&^_V3IBcES?a*RGHK5yTR#W7i%R8QcZ_=IhkZ*;@7&LIhpcH|BE+CAc3n zCrZ`Y?SSvPa%W=s>}X5=)0H_MibSNH2Ti{Uy}r#M<@B{c;5!>0XqW+@*zEy1qJq5K z`TN{qu-IzsvmsJoB3_=W;7wc_^R8fmNxtn(bw6dV8`(0%Q%m62%ufs=H^GiNv(p|p zanow+we>mD%SVRrKE51OzC8879N-23B>ittztM3X9w0Qx*So`CQmO~pi{aDh- zBa^o8@t$6)Our#%H}lZ+(uZ8Od_~si(aoVBF>f7oIzIgc%7M8dRq(5LoxQ^5PYOQx zL!WE1=ox*LF=g)+n-$Hz%gQ`n(U$O}$INOp&j9Mn0^8p#Q>K*3u zy%Ydx12C!6n7^->t=e1mM%- zwo&3Tz!}PW-v2|Lf89XtJr<-bPoPXk_x$fd!{;@yYfLcp<{jtAw}P8<8Rix< zU{1H9KqY9MHqcm*AEkIxd?(Nl=gy*mi;J#6TqkD@bx}AdRZEF1KRYYH<-S_Tuh8i4 z{T3Ou*Z^3p5&6#%V(xM}=+Cti*kn@T9)v?a5m&IPIUN6-r}2xZ>Cxf5IltiF^`XMK zh39F#5F1!4rtgNk&fpc>+TWDmgzuFW?M#5wH|L0`Qah^nOP~Q>k+%$MPPSP-Wz6hd z!931Rd=X<$hx~+Y?Q$yDf(9?WQlE-TN9)U*w@4E&{;uohIiL7V{$_`Vov#_eOB)Ol zJ}BL`gRKr25Ycp~Yc3fUcMN``yweAULPHm?VMNpB+`m}*A142cz-k$|;c#(!D_2RQ z_vql7=0)co$)*1Nmn?gZS^cObdI_;0`MOOIdcBFVC|F$rWs4K^NuCSD+#bCca{0`c zt91s#&_DJHFE#?*=CQszk$a$%=GQMbNxH_Q*6h+qo<_ne<%-_2AJvqa^zqB&9Wt8t zG50YQ>mE<8n2s(GFri!fB?{(Rvq>9w+l+FD(BTwK+Dw){t}GggNS^c6z6 zGdq1O=N!fLAA%e>!DBBLPh`$X3l**!R`pVnp+l{j%A<}{hu=a zFEzmA#GPfDVJX2*!_O^xBmBeIj-S=-96`0GcHA>Og>~OPMIvtFk33PeZ@C8z-b0lM zwIUy78kkh4MJ*I&#^YSNcikb2TRY))0dNT&56L(aX5#5lE|o0W+yb8*a7=3&n=(%m8p&uIGA%Hr-WZ_jLWmu@tIhS zxl)ZC*SB9TXFw62CLA@$e;;7>8>+oW10Z8t^+v|{ZJ!v?7> zb5$9Sl1s89O;s-~Pv-c4Opk)Dh8e^GR0+)^UZHS936YWLfX_`Wk7CEKh2UJAAyxF7 zI70_`OC*TP{u1KF*3h>?f>`ACL$r*PA>N0@utz2(<60yM@4}c7-xox!XHVe7^`Mca zO}SL$e8%j@&1|IxYJlvNb1K$Hmmr2v_#5t!ex^=lSYHOtaD9A>tyDx!okf<~3MVM> zPUU%yF2ZVVvCZtHLxrL*U%TqRw|OaMoKC@ygIkO^%+ry{O+(o*dKb3dkt7UlR!woj z0x-AZ;sRT}zV1HDaQ6fVczpo$_qOvq)5flh)5JaF5hc3>U3F_R&M={ELo|FovLfG` z#i{*ypL+1+U1O1&v}<)#hNEttNUkGN`*(R!tXn#fR?L^u>r2a%*-dnB#fjU6G2*J~ zvlmAb8VB^EP<&)FzvQxI-1UKO`ZTe_J|)J&E<*16w)4soFef2&!m}o$yvwhCbSiCD zpdK-Y4hb1wB>ZpRX?GquJWzXG?AE)iQPW@}M%d`&i8kxrSKadI#3o#S0|uqT5fh%) znG_us0L}N{%36thWV~12k$*EBeP-iUp0ad7@G1!?G3@ScxV^fZ$-7^aeTWWVYx^Tr z{uiM%?C7_Gaj8e`83W&56~C@k`yk3VjdS;!HmRNo3qUDgw11UR(vP$AiA~m_@0&Ne z(}3lXdJi1m6H;{W-4)$2A!TZ_X)@$9-(dp{tnUk&m%iCKbaDK{jgG0ViN-Hewth5N zO8*}HYRA=WdikXGd-@$v@&4mJU&rj}&Vm%D-gS*C7p*bkw$5J6?P2@G(w=WmZlNe# zHbX$#vwmXq+nte~btdV^#yz?ppA1=#qtdev_OpGh=Jj4T$7E}u(1E3c} zK-Ueb)iaxReO%#~X5l_*;=)G93Ge{L|LeL4uNGft6r2zwozFH$Mb(9#^Cs|T2KFtsQ;>^K=BbaLJqrw>( zvAuZTc_;R%P1+E*9_!0|!j2VXECKmrIJauy=rJ;L+tTq9TZj8BFGvBuzZMjSe-=4< zLu;#~)3|q;F zT{((v6VVz)QZFYCoSSK)BjL67N{qO)Gcf=;wq_C5(D|)_$?eI@RpfaCd2xbqqV*#{A^v2ds>kZ|yaAy_*a|^l`3=-&9knnRANlsd6Pp1e3FU z_pPm!$VkLZ;_Zu*8d46vRjEAbGh>G%XjR4yqdPtiOMBZr{;EpVu;3di7XJq|Qk* zIXt4RUpCOYhI6qq(VyQG)H7i%5Pc~4FjXF=bmEDnu|l)>SxxVo-z;Mhgu$Sn?8gRc ztsOj0C!@)QN@XuHhtpYsGOVf7?Z55C!Wy({dvL2dVa38)@k^^&?B&*2Mk*fM;H#(A z?04yqLY$(|_!{4IE??~|Zc@3$j&V<=TVGFWt$^cyhG^}VQxg~w@ehL6W_~#4Vw2U+ zA{(Z*JXr>QZZCdaq;ar#Bcqzf{<9q6zg(rj2tseZ$(DP-9weba>D97qhW6MB@vfqqN`5qDh$FAx-iG~Qll;8qwo_h)jB3? zc}=_apaU37X^b6?cHEMFIg=$LnC-%~Jev$mOIIDzcKbCUDgY4%u-f^ClvE_4N{3$z zT_vL}7^`WJ9cV~oen!EMrEGP(&6*B2K z@r@2^^`|K@k!}DXt_uVy1~@!I;!{K6=5y)5PQ(ysncYo8IBI@c$ap?CwJx6^_r0=8iu9u)L z3Ls$C6Ie9FHtJ}T0qiiCz|>!t`ZdWx86g%)U4E^w#H*V#Qjlm~2&AP}lcIK*KL@%13HLd>3Ht`M_$Q zY3yRU`gmLz=e7GQcKL{AZ+hwlVCbGiLu7iCm~#-nl+YsqQQeE=s$nni9Ovrm3bK^aMM6{in($azFz?dGSpS8&w-O&5-mMR#`-SqG}Si3inw=Zq1FQQKZ) z(VTR$W0p_gg&`^O0s+(}VN7R~~JgtX@^ed?}ez2&_NijNOcUes{sc?msb1)!jPYG&rmGt&8Bd7Z&GCX39@gp9j_% zJa`L7QC?fPS&zwh%Zs(8LuukRTqox`@~mw?+t&dBZ1B$##Js{Z0_CQ^D~_0Gr4fSa zTWGRELbnbAi$l_?dcvp)^Z>R0I*3p{A)C=VXg%yfVtYXYxkerjd3s$ci7qsL?uJsW zA_iw{F1O+9?jzd1z5 zGEP0hw34^=_r7Ul#;`xUz+YFOJ~+3ZT{1Q=|In`~^|E#Bu;r6tOVqq0`I-;Fl-C8W}kTU)Kyf+Iy ze?P(&MeL@)v4mhe1r5s3utRlQ;_T~2Kz4!;ZOcYadP}@7B%8$Y9^D6cKrz6&J=d}h zs}Y$b<7)G~_qCTp;cHN#B{K5MfIvD*>P*j%RAwn~`dF+C>S|2cVz27YMP@wi#Dbv;Iy+NoeS<*fD{z4!b*)il9Ssj820|)8N!hMWa7lxos0tYSlNnQI!vs`Ga9tJw<+Jepf%OKV! zYzNl5yEL({E==)vh^V~R2%tGwo&V>lCSULum7au?}n;_+ff65EXmdf!_tHP~% zOPg9}pvl*^2Cd>tH=FFH`ce!iaNwTCMC@t>AAKt9oGmCmvg2~TJhaT?)qb=3be;}; p?T_PM-nyRu2|nt+9saKA|4=bVnwK$&{(A%)tbeh(yw>Zt{{k;cUjP6A literal 0 HcmV?d00001 diff --git a/start.sh b/start.sh new file mode 100644 index 0000000..0e9c72b --- /dev/null +++ b/start.sh @@ -0,0 +1,129 @@ +#!/bin/bash +set -eu + +# ============================================================================ +# LiteLLM Cloudron Startup Script +# Configures LiteLLM with Cloudron's PostgreSQL and Redis addons +# ============================================================================ + +echo "=== LiteLLM Cloudron Startup ===" + +# --- Cloudron PostgreSQL addon provides: +# CLOUDRON_POSTGRESQL_URL - postgresql://user:pass@host:port/dbname +# CLOUDRON_POSTGRESQL_USERNAME +# CLOUDRON_POSTGRESQL_PASSWORD +# CLOUDRON_POSTGRESQL_HOST +# CLOUDRON_POSTGRESQL_PORT +# CLOUDRON_POSTGRESQL_DATABASE + +# --- Cloudron Redis addon provides: +# CLOUDRON_REDIS_URL - redis://user:pass@host:port +# CLOUDRON_REDIS_USERNAME +# CLOUDRON_REDIS_PASSWORD +# CLOUDRON_REDIS_HOST +# CLOUDRON_REDIS_PORT + +# --- Cloudron provides: +# CLOUDRON_APP_DOMAIN - the app's domain (e.g. gateway.inference.coop) +# CLOUDRON_API_ORIGIN - origin for API calls +# CLOUDRON_WEB_ORIGIN - origin for web UI + +# --- LiteLLM required env vars --- + +# Generate a master key if not already in /app/data +MASTER_KEY_FILE="/app/data/.master_key" +if [ ! -f "$MASTER_KEY_FILE" ]; then + echo "Generating master key..." + python3 -c "import secrets; print('sk-' + secrets.token_urlsafe(32))" > "$MASTER_KEY_FILE" +fi +export LITELLM_MASTER_KEY=$(cat "$MASTER_KEY_FILE") + +# Generate a salt key if not already in /app/data +SALT_KEY_FILE="/app/data/.salt_key" +if [ ! -f "$SALT_KEY_FILE" ]; then + echo "Generating salt key..." + python3 -c "import secrets; print(secrets.token_urlsafe(32))" > "$SALT_KEY_FILE" +fi +export LITELLM_SALT_KEY=$(cat "$SALT_KEY_FILE") + +# Use Cloudron's PostgreSQL +export DATABASE_URL="${CLOUDRON_POSTGRESQL_URL}" + +# Use Cloudron's Redis (for rate limiting, caching) +export REDIS_HOST="${CLOUDRON_REDIS_HOST}" +export REDIS_PORT="${CLOUDRON_REDIS_PORT}" +export REDIS_PASSWORD="${CLOUDRON_REDIS_PASSWORD}" + +# Store models in DB (manage from Admin UI) +export STORE_MODEL_IN_DB="True" + +# Don't run schema migrations on every start — let LiteLLM handle it +# (for single-instance deployment this is fine) +export DISABLE_SCHEMA_UPDATE="false" + +# UI configuration — use Cloudron's domain +export UI_USERNAME="admin" +export UI_PASSWORD="${LITELLM_MASTER_KEY}" + +# Proxy settings for Cloudron's reverse proxy +export LITELLM_PROXY_BASE_URL="https://${CLOUDRON_APP_DOMAIN}" + +# Allow requests from Cloudron's Open WebUI, LobeChat, etc. +export LITELLM_CORS_ALLOWED_ORIGINS="${CLOUDRON_WEB_ORIGIN:-*}" + +echo "PostgreSQL: ${CLOUDRON_POSTGRESQL_HOST}:${CLOUDRON_POSTGRESQL_PORT}" +echo "Redis: ${CLOUDRON_REDIS_HOST}:${CLOUDRON_REDIS_PORT}" +echo "Domain: ${CLOUDRON_APP_DOMAIN}" + +# --- Write config.yaml with Cloudron provider keys if set --- + +CONFIG_FILE="/app/data/config.yaml" + +if [ ! -f "$CONFIG_FILE" ] || [ ! -s "$CONFIG_FILE" ]; then + echo "Writing default config.yaml..." + cat > "$CONFIG_FILE" << 'YAML' +# LiteLLM config for inference.coop +# MVP config: Tinfoil only (TEE-protected, privacy-first) +# Two models: DeepSeek V4 Flash (default, best agents) + GPT-OSS 120B (cheapest) +# Provider key set via environment variable or Admin UI + +model_list: + # DeepSeek V4 Flash — default model + - model_name: deepseek-v4-flash + litellm_params: + model: openai/deepseek-v4-flash + api_base: https://api.tinfoil.sh/v1 + api_key: os.environ/TINFOIL_API_KEY + + # GPT-OSS 120B — lightweight fallback + - model_name: gpt-oss-120b + litellm_params: + model: openai/gpt-oss-120b + api_base: https://api.tinfoil.sh/v1 + api_key: os.environ/TINFOIL_API_KEY + +# Fallback: if DeepSeek is down, use GPT-OSS +router_settings: + num_retries: 2 + timeout: 30 + fallbacks: + - deepseek-v4-flash: + - gpt-oss-120b + +general_settings: + master_key: os.environ/LITELLM_MASTER_KEY + database_url: os.environ/DATABASE_URL + store_model_in_db: true + +litellm_settings: + salt_key: os.environ/LITELLM_SALT_KEY + drop_params: true + num_threads: 4 + request_timeout: 30 +YAML + echo "Config written to $CONFIG_FILE" +fi + +# --- Start LiteLLM --- +echo "Starting LiteLLM on port 4000..." +exec litellm --config "$CONFIG_FILE" --port 4000 --host 0.0.0.0 \ No newline at end of file