From e05fec0ca81ccbcc1dda5612859275e8c2077a73 Mon Sep 17 00:00:00 2001 From: inference-bot Date: Wed, 9 Sep 2026 13:21:01 -0600 Subject: [PATCH] Switch to Tinfoil (TEE): add tinfoil-proxy sidecar for EHBP encryption, point LiteLLM at local proxy --- Dockerfile | 10 ++++++++++ config.yaml | 22 ++++++++++++---------- start.sh | 13 +++++++++++++ 3 files changed, 35 insertions(+), 10 deletions(-) diff --git a/Dockerfile b/Dockerfile index c1a3265..b292c5c 100644 --- a/Dockerfile +++ b/Dockerfile @@ -8,6 +8,16 @@ FROM ghcr.io/berriai/litellm:main-v1.74.0-stable # Create the data directory (Cloudron mounts the localstorage volume here) RUN mkdir -p /app/data +# Tinfoil proxy — a verified local proxy that encrypts request/response bodies +# with EHBP (HPKE) before forwarding to the Tinfoil enclave. LiteLLM's openai/ +# provider sends plaintext, which Tinfoil rejects (426 EHBP_REQUIRED), so the +# proxy sits between LiteLLM and the enclave. Statically-linked Go binary, +# pinned to a specific release for reproducible builds. Downloaded with Python +# (guaranteed present in the Python base image) rather than curl/wget. +ARG TINFOIL_PROXY_VERSION=v0.2.3 +RUN python3 -c "import urllib.request; urllib.request.urlretrieve('https://github.com/tinfoilsh/tinfoil-proxy/releases/download/${TINFOIL_PROXY_VERSION}/tinfoil-proxy-linux-amd64', '/app/code/tinfoil-proxy')" \ + && chmod +x /app/code/tinfoil-proxy + # Create a startup script that configures LiteLLM with Cloudron addons COPY start.sh /app/code/start.sh RUN chmod +x /app/code/start.sh diff --git a/config.yaml b/config.yaml index 56d3104..c7acf49 100644 --- a/config.yaml +++ b/config.yaml @@ -1,22 +1,24 @@ # LiteLLM config for inference.coop -# TEST config: Ollama Cloud (existing subscription) — plumbing smoke test -# Uses Ollama Cloud's OpenAI-compatible endpoint (/v1) via LiteLLM's openai/ provider. -# Provider key set via environment variable OLLAMA_API_KEY. +# PRODUCTION config: Tinfoil (TEE-protected inference) +# LiteLLM talks plaintext OpenAI to a local Tinfoil proxy (127.0.0.1:3301), +# which verifies the enclave attestation and encrypts request/response bodies +# with EHBP (HPKE) before forwarding to the Tinfoil enclave. Provider key set +# via environment variable TINFOIL_API_KEY. model_list: - # DeepSeek V4 Flash — default model + # DeepSeek V4 Flash — default model (1M context, tool calling) - model_name: deepseek-v4-flash litellm_params: - model: openai/deepseek-v4-flash:0731 - api_base: https://ollama.com/v1 - api_key: os.environ/OLLAMA_API_KEY + model: openai/deepseek-v4-flash + api_base: http://127.0.0.1:3301/v1 + api_key: os.environ/TINFOIL_API_KEY # GPT-OSS 120B — lightweight fallback - model_name: gpt-oss-120b litellm_params: - model: openai/gpt-oss:120b - api_base: https://ollama.com/v1 - api_key: os.environ/OLLAMA_API_KEY + model: openai/gpt-oss-120b + api_base: http://127.0.0.1:3301/v1 + api_key: os.environ/TINFOIL_API_KEY # Fallback: if DeepSeek is down, use GPT-OSS router_settings: diff --git a/start.sh b/start.sh index 0e9c72b..11c7796 100644 --- a/start.sh +++ b/start.sh @@ -126,4 +126,17 @@ fi # --- Start LiteLLM --- echo "Starting LiteLLM on port 4000..." + +# Start the Tinfoil proxy as a local sidecar. It verifies the enclave +# attestation and encrypts request/response bodies with EHBP (HPKE), so +# LiteLLM can talk plaintext OpenAI to it while the proxy handles the +# end-to-end encryption to the Tinfoil enclave. +echo "Starting Tinfoil proxy on 127.0.0.1:3301..." +/app/code/tinfoil-proxy -b 127.0.0.1 -p 3301 & +TINFOIL_PROXY_PID=$! +echo "Tinfoil proxy PID: $TINFOIL_PROXY_PID" + +# Give the proxy a moment to verify the enclave attestation +sleep 3 + exec litellm --config "$CONFIG_FILE" --port 4000 --host 0.0.0.0 \ No newline at end of file