diff --git a/deploy/open-webui/README.md b/deploy/open-webui/README.md new file mode 100644 index 0000000000000000000000000000000000000000..0f1317f745f31530fa687c8a35e5e3a726d1dda2 --- /dev/null +++ b/deploy/open-webui/README.md @@ -0,0 +1,71 @@ +# Open WebUI on Fly.io (minimal, OpenRouter) + +You do **not** need to host your own model. [OpenRouter](https://openrouter.ai/) runs the LLMs; this Fly app only runs the Open WebUI shell. + +### Disk vs RAM + +The container image still includes ML libraries on **disk** (upstream wheels). These settings target **runtime memory**: chat and API proxy use remote inference, and `RAG_EMBEDDING_ENGINE=openai` plus `AUDIO_STT_ENGINE=openai` tell Open WebUI to call your configured OpenAI-compatible API for embeddings and speech-to-text instead of loading local SentenceTransformers / Whisper models into RAM (see Open WebUI [performance](https://docs.openwebui.com/troubleshooting/performance/) notes on embedding offload). + +You may still see higher RSS if you use features that pull in other local code paths. For a fresh volume, the `[env]` values apply on first boot; if you already ran Open WebUI, stored admin settings can override some env vars ([PersistentConfig](https://docs.openwebui.com/reference/env-configuration/) — adjust in Admin or reset the volume for a clean slate). + +## Prerequisites + +- [Fly CLI](https://fly.io/docs/hands-on/install-flyctl/) and `fly auth login` +- An OpenRouter API key (`sk-or-…` from [openrouter.ai/keys](https://openrouter.ai/keys)) + +## One-time setup + +1. Edit `fly.toml` and set `app = "your-unique-name"` (globally unique on Fly). + +2. Create the app (if it does not exist): + + ```bash + fly apps create your-unique-name + ``` + +3. Create a volume for SQLite and uploads (region must match `primary_region` in `fly.toml`): + + ```bash + fly volumes create open_webui_data --region iad --size 3 + ``` + +4. Set secrets: + + ```bash + fly secrets set OPENAI_API_KEY="sk-or-..." \ + WEBUI_SECRET_KEY="$(openssl rand -hex 32)" \ + WEBUI_URL="https://your-unique-name.fly.dev" + ``` + + Optional but convenient: create the first admin in one step (disables open signup on first boot): + + ```bash + fly secrets set WEBUI_ADMIN_EMAIL="you@example.com" WEBUI_ADMIN_PASSWORD='strong-password-here' + ``` + +5. Deploy: + + ```bash + cd deploy/open-webui && fly deploy + ``` + +Open `https://your-unique-name.fly.dev`. In the UI, pick a model served by OpenRouter (IDs like `openai/gpt-4o`, `anthropic/claude-3.5-sonnet`, etc.). + +## If the machine runs out of memory + +The `fly.toml` env is tuned to avoid local embedding/STT models in RAM; if it still OOMs (large chats, document uploads, many users), scale up: + +```bash +fly scale memory 2048 +# or 4096 if needed +``` + +## When you *would* self-host a model + +Only if you want **local / private inference** (no third-party API). That usually means **Ollama** or similar on a GPU-capable host, not this minimal Fly setup. For “just chat via API,” OpenRouter (or direct OpenAI) is enough. + +## Notes + +- `OPENAI_API_BASE_URL` points at OpenRouter’s OpenAI-compatible API; `OPENAI_API_KEY` is your OpenRouter key. +- After the first run, some settings are stored in the volume under `/app/backend/data` (see Open WebUI docs on `PersistentConfig`). +- Pin the image tag instead of `:main` in `fly.toml` if you want reproducible deploys. diff --git a/deploy/open-webui/fly.toml b/deploy/open-webui/fly.toml new file mode 100644 index 0000000000000000000000000000000000000000..684ae6d108e6a90906ef95c5075b468fe8280134 --- /dev/null +++ b/deploy/open-webui/fly.toml @@ -0,0 +1,59 @@ +# Open WebUI on Fly.io — OpenRouter only (no Ollama). +# Runtime RAM: RAG_EMBEDDING_ENGINE + AUDIO_STT_ENGINE use your OpenRouter HTTP API so local +# SentenceTransformers / Whisper weights are not loaded into memory (torch may still exist on disk in the image). +# Replace `app` with your Fly app name, then: +# fly volumes create open_webui_data --region --size 3 +# fly secrets set OPENAI_API_KEY=sk-or-... WEBUI_SECRET_KEY=$(openssl rand -hex 32) +# fly secrets set WEBUI_URL=https://.fly.dev +# Optional (recommended): headless admin, signup disabled automatically: +# fly secrets set WEBUI_ADMIN_EMAIL=you@example.com WEBUI_ADMIN_PASSWORD='...' +# fly deploy + +app = "open-webui" +primary_region = "iad" + +[build] + image = "ghcr.io/open-webui/open-webui:main" + +[env] + OPENAI_API_BASE_URL = "https://openrouter.ai/api/v1" + ENABLE_OLLAMA_API = "false" + # One worker so we do not duplicate in-RAM state (see Open WebUI scaling docs). + UVICORN_WORKERS = "1" + # Remote embeddings via OpenAI-compatible API (OpenRouter) — avoids ~500MB+ local embedding model in RAM. + RAG_EMBEDDING_ENGINE = "openai" + # Remote speech-to-text when using voice — avoids loading faster-whisper into RAM. + AUDIO_STT_ENGINE = "openai" + # Speeds up model list with OpenRouter’s large catalog (small in-memory cache). + ENABLE_BASE_MODELS_CACHE = "true" + +[[mounts]] + source = "open_webui_data" + destination = "/app/backend/data" + +[[services]] + internal_port = 8080 + protocol = "tcp" + + [[services.ports]] + port = 80 + handlers = ["http"] + force_https = true + + [[services.ports]] + port = 443 + handlers = ["tls", "http"] + + [services.concurrency] + type = "connections" + hard_limit = 250 + soft_limit = 200 + + [[services.http_checks]] + interval = "15s" + timeout = "5s" + grace_period = "60s" + method = "GET" + path = "/health" + protocol = "http" + tls_skip_verify = false