Files
inference-co-op-bot e8c7a55933 Add upgrade runbook + bake memory limit and LITELLM_MIGRATION_DIR into packaging
The 2026-10-02 outage had three causes the runbook now prevents:
- floating tag drift (README now mandates exact-tag pinning)
- 256MB default cgroup cap OOM-killing the prisma migration engine
  (manifest now sets memoryLimit: 2147483648)
- no _prisma_migrations ledger -> P3005 crash-loop on boot
  (start.sh now exports LITELLM_MIGRATION_DIR=/app/data/migrations)

README 'Updating LiteLLM' section rewritten from the old (wrong)
floating-tag procedure into a runbook with rules + verify checklist.
2026-10-02 00:52:50 -06:00

175 lines
6.3 KiB
Bash

#!/bin/bash
set -eu
# ============================================================================
# LiteLLM Cloudron Startup Script
# Configures LiteLLM with Cloudron's PostgreSQL and Redis addons
# ============================================================================
echo "=== LiteLLM Cloudron Startup ==="
# --- Cloudron PostgreSQL addon provides:
# CLOUDRON_POSTGRESQL_URL - postgresql://user:pass@host:port/dbname
# CLOUDRON_POSTGRESQL_USERNAME
# CLOUDRON_POSTGRESQL_PASSWORD
# CLOUDRON_POSTGRESQL_HOST
# CLOUDRON_POSTGRESQL_PORT
# CLOUDRON_POSTGRESQL_DATABASE
# --- Cloudron Redis addon provides:
# CLOUDRON_REDIS_URL - redis://user:pass@host:port
# CLOUDRON_REDIS_USERNAME
# CLOUDRON_REDIS_PASSWORD
# CLOUDRON_REDIS_HOST
# CLOUDRON_REDIS_PORT
# --- Cloudron provides:
# CLOUDRON_APP_DOMAIN - the app's domain (e.g. gateway.inference.coop)
# CLOUDRON_API_ORIGIN - origin for API calls
# CLOUDRON_WEB_ORIGIN - origin for web UI
# --- LiteLLM required env vars ---
# Generate a master key if not already in /app/data
MASTER_KEY_FILE="/app/data/.master_key"
if [ ! -f "$MASTER_KEY_FILE" ]; then
echo "Generating master key..."
python3 -c "import secrets; print('sk-' + secrets.token_urlsafe(32))" > "$MASTER_KEY_FILE"
fi
export LITELLM_MASTER_KEY=$(cat "$MASTER_KEY_FILE")
# Generate a salt key if not already in /app/data
SALT_KEY_FILE="/app/data/.salt_key"
if [ ! -f "$SALT_KEY_FILE" ]; then
echo "Generating salt key..."
python3 -c "import secrets; print(secrets.token_urlsafe(32))" > "$SALT_KEY_FILE"
fi
export LITELLM_SALT_KEY=$(cat "$SALT_KEY_FILE")
# Use Cloudron's PostgreSQL
export DATABASE_URL="${CLOUDRON_POSTGRESQL_URL}"
# Use Cloudron's Redis (for rate limiting, caching)
export REDIS_HOST="${CLOUDRON_REDIS_HOST}"
export REDIS_PORT="${CLOUDRON_REDIS_PORT}"
export REDIS_PASSWORD="${CLOUDRON_REDIS_PASSWORD}"
# Store models in DB (manage from Admin UI)
export STORE_MODEL_IN_DB="True"
# Don't run schema migrations on every start — let LiteLLM handle it
# (for single-instance deployment this is fine)
export DISABLE_SCHEMA_UPDATE="false"
# The DB was created by `prisma db push` (no _prisma_migrations ledger), so on
# first boot after a version change LiteLLM hits P3005 ("schema is not empty")
# and must baseline the existing schema. That baseline write needs a WRITABLE
# directory — the package dir is read-only under Cloudron's rootfs, so point it
# at the persistent volume. Without this the boot migration crash-loops.
export LITELLM_MIGRATION_DIR="/app/data/migrations"
# UI configuration — use Cloudron's domain
export UI_USERNAME="admin"
export UI_PASSWORD="${LITELLM_MASTER_KEY}"
# DISABLE the admin UI and API docs. The gateway must stay publicly reachable
# for the /v1 API (member keys), but we don't want to advertise a login-gated
# admin console (/ui) or LiteLLM's Swagger/ReDoc/openapi surfaces. Admin
# visibility comes from the admin dashboard / the API directly.
export DISABLE_ADMIN_UI="True"
export NO_DOCS="True"
export NO_REDOC="True"
export NO_OPENAPI="True"
# Proxy settings for Cloudron's reverse proxy
export LITELLM_PROXY_BASE_URL="https://${CLOUDRON_APP_DOMAIN}"
# CORS: LiteLLM v1.84.0 reads LITELLM_CORS_ORIGINS natively (upstream added
# the env var in place of the old hardcoded `origins = ["*"]`). Lock browser
# access to the co-op's own chat origin. The gateway is called
# server-to-server (chat → portal → gateway) and by member API keys
# (curl/SDKs, not browsers), so no wildcard is needed. A `*` here would let
# any website's JS hit the gateway.
export LITELLM_CORS_ORIGINS="https://chat.inference.coop"
echo "PostgreSQL: ${CLOUDRON_POSTGRESQL_HOST}:${CLOUDRON_POSTGRESQL_PORT}"
echo "Redis: ${CLOUDRON_REDIS_HOST}:${CLOUDRON_REDIS_PORT}"
echo "Domain: ${CLOUDRON_APP_DOMAIN}"
# --- Write config.yaml with Cloudron provider keys if set ---
CONFIG_FILE="/app/data/config.yaml"
if [ ! -f "$CONFIG_FILE" ] || [ ! -s "$CONFIG_FILE" ]; then
echo "Writing default config.yaml..."
cat > "$CONFIG_FILE" << 'YAML'
# LiteLLM config for inference.coop
# MVP config: Tinfoil only (TEE-protected, privacy-first)
# Two models: DeepSeek V4 Flash (default, best agents) + GPT-OSS 120B (cheapest)
# Provider key set via environment variable or Admin UI
model_list:
# DeepSeek V4 Flash — default model
- model_name: deepseek-v4-flash
litellm_params:
model: openai/deepseek-v4-flash
api_base: https://api.tinfoil.sh/v1
api_key: os.environ/TINFOIL_API_KEY
# GPT-OSS 120B — lightweight fallback
- model_name: gpt-oss-120b
litellm_params:
model: openai/gpt-oss-120b
api_base: https://api.tinfoil.sh/v1
api_key: os.environ/TINFOIL_API_KEY
# Fallback: if DeepSeek is down, use GPT-OSS
router_settings:
num_retries: 2
timeout: 30
fallbacks:
- deepseek-v4-flash:
- gpt-oss-120b
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
database_url: os.environ/DATABASE_URL
store_model_in_db: true
litellm_settings:
salt_key: os.environ/LITELLM_SALT_KEY
drop_params: true
num_threads: 4
request_timeout: 30
YAML
echo "Config written to $CONFIG_FILE"
fi
# --- Start LiteLLM ---
echo "Starting LiteLLM on port 4000..."
# Start the Tinfoil proxy as a local sidecar. It verifies the enclave
# attestation and encrypts request/response bodies with EHBP (HPKE), so
# LiteLLM can talk plaintext OpenAI to it while the proxy handles the
# end-to-end encryption to the Tinfoil enclave.
#
# The proxy binary is downloaded at runtime into /app/data (persistent,
# writable) because Cloudron's build sandbox has no outbound network access to
# GitHub. The running container does, so we fetch it here on first start.
TINFOIL_PROXY_BIN="/app/data/tinfoil-proxy"
TINFOIL_PROXY_VERSION="v0.2.3"
if [ ! -f "$TINFOIL_PROXY_BIN" ]; then
echo "Downloading Tinfoil proxy ${TINFOIL_PROXY_VERSION}..."
python3 -c "import urllib.request; urllib.request.urlretrieve('https://github.com/tinfoilsh/tinfoil-proxy/releases/download/${TINFOIL_PROXY_VERSION}/tinfoil-proxy-linux-amd64', '${TINFOIL_PROXY_BIN}')"
chmod +x "$TINFOIL_PROXY_BIN"
fi
echo "Starting Tinfoil proxy on 127.0.0.1:3301..."
"$TINFOIL_PROXY_BIN" -b 127.0.0.1 -p 3301 &
TINFOIL_PROXY_PID=$!
echo "Tinfoil proxy PID: $TINFOIL_PROXY_PID"
# Give the proxy a moment to verify the enclave attestation
sleep 3
exec litellm --config "$CONFIG_FILE" --port 4000 --host 0.0.0.0