Add Windows (PowerShell) support
- scripts/setup.ps1, scripts/pull-models.ps1, run.ps1 — PowerShell equivalents of the bash setup/pull/run scripts - README + docs/DEPLOY.md: Windows quickstart and Task Scheduler autostart Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
9938d46a67
commit
95da8fb38d
@@ -61,6 +61,13 @@ git clone <your-fork-url> llm-router && cd llm-router
|
|||||||
./run.sh
|
./run.sh
|
||||||
```
|
```
|
||||||
|
|
||||||
|
On **Windows** (PowerShell), use the `.ps1` equivalents:
|
||||||
|
```powershell
|
||||||
|
powershell -ExecutionPolicy Bypass -File .\scripts\setup.ps1
|
||||||
|
powershell -ExecutionPolicy Bypass -File .\scripts\pull-models.ps1
|
||||||
|
powershell -ExecutionPolicy Bypass -File .\run.ps1
|
||||||
|
```
|
||||||
|
|
||||||
Then point any OpenAI client at `http://localhost:8080/v1` with `model: "auto"`:
|
Then point any OpenAI client at `http://localhost:8080/v1` with `model: "auto"`:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
|
|||||||
@@ -27,6 +27,27 @@ launchctl kickstart -k gui/$(id -u)/com.llm-router
|
|||||||
systemctl --user restart llm-router
|
systemctl --user restart llm-router
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Windows
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
powershell -ExecutionPolicy Bypass -File .\scripts\setup.ps1 # venv + LiteLLM
|
||||||
|
powershell -ExecutionPolicy Bypass -File .\scripts\pull-models.ps1 # pull models
|
||||||
|
powershell -ExecutionPolicy Bypass -File .\run.ps1 # LiteLLM + router
|
||||||
|
```
|
||||||
|
`run.ps1` starts LiteLLM in the background, waits for it, then runs the router in
|
||||||
|
the foreground; closing it stops both.
|
||||||
|
|
||||||
|
**Run at login (Task Scheduler):**
|
||||||
|
```powershell
|
||||||
|
$action = New-ScheduledTaskAction -Execute "powershell.exe" `
|
||||||
|
-Argument "-ExecutionPolicy Bypass -WindowStyle Hidden -File `"$PWD\run.ps1`""
|
||||||
|
$trigger = New-ScheduledTaskTrigger -AtLogOn
|
||||||
|
Register-ScheduledTask -TaskName "llm-router" -Action $action -Trigger $trigger `
|
||||||
|
-Settings (New-ScheduledTaskSettingsSet -RestartCount 3 -RestartInterval (New-TimeSpan -Minutes 1))
|
||||||
|
```
|
||||||
|
Requires **Python 3.12** (`winget install Python.Python.3.12`) and Ollama for
|
||||||
|
Windows. The router (stdlib Python) and LiteLLM run the same as on macOS/Linux.
|
||||||
|
|
||||||
## Networking
|
## Networking
|
||||||
|
|
||||||
- The router binds `ROUTER_HOST` (default `0.0.0.0` = reachable on the LAN).
|
- The router binds `ROUTER_HOST` (default `0.0.0.0` = reachable on the LAN).
|
||||||
|
|||||||
@@ -0,0 +1,39 @@
|
|||||||
|
# Serve on Windows: LiteLLM (:4000) + router (:8080). Windows equivalent of run.sh.
|
||||||
|
# Run: powershell -ExecutionPolicy Bypass -File .\run.ps1
|
||||||
|
$ErrorActionPreference = "Stop"
|
||||||
|
Set-Location $PSScriptRoot
|
||||||
|
|
||||||
|
# load keys from files into env (kept out of git)
|
||||||
|
if (Test-Path ".apikey") { $env:ROUTER_API_KEY = (Get-Content ".apikey" -Raw).Trim() }
|
||||||
|
if (Test-Path ".uncensored_key") { $env:ROUTER_UNCENSORED_KEY = (Get-Content ".uncensored_key" -Raw).Trim() }
|
||||||
|
if (-not $env:OLLAMA_KEEP_ALIVE) { $env:OLLAMA_KEEP_ALIVE = "30m" }
|
||||||
|
|
||||||
|
# warm the small classifier so ambiguous routing has no cold start (best effort)
|
||||||
|
try {
|
||||||
|
Invoke-RestMethod -Method Post -Uri "http://127.0.0.1:11434/api/generate" -ContentType "application/json" `
|
||||||
|
-TimeoutSec 10 -Body '{"model":"qwen3:8b","prompt":"ok","stream":false,"keep_alive":-1}' | Out-Null
|
||||||
|
} catch {}
|
||||||
|
|
||||||
|
# LiteLLM backend (localhost only)
|
||||||
|
Write-Host ">> starting LiteLLM on 127.0.0.1:4000 ..."
|
||||||
|
$litellm = Start-Process -FilePath ".\.venv\Scripts\litellm.exe" `
|
||||||
|
-ArgumentList "--config", "litellm.config.yaml", "--host", "127.0.0.1", "--port", "4000" `
|
||||||
|
-RedirectStandardOutput "litellm.log" -RedirectStandardError "litellm.err.log" -NoNewWindow -PassThru
|
||||||
|
$litellm.Id | Set-Content ".litellm.pid"
|
||||||
|
|
||||||
|
for ($i = 0; $i -lt 60; $i++) {
|
||||||
|
try { Invoke-WebRequest "http://127.0.0.1:4000/health/liveliness" -TimeoutSec 2 -UseBasicParsing | Out-Null; break }
|
||||||
|
catch { Start-Sleep -Seconds 2 }
|
||||||
|
}
|
||||||
|
|
||||||
|
# Router (network-facing). Trap tears down LiteLLM on exit.
|
||||||
|
$env:ROUTER_UPSTREAM = "http://127.0.0.1:4000/v1"
|
||||||
|
if (-not $env:ROUTER_HOST) { $env:ROUTER_HOST = "0.0.0.0" }
|
||||||
|
Write-Host ">> router on http://$($env:ROUTER_HOST):8080/v1 (use model 'auto')"
|
||||||
|
try {
|
||||||
|
& ".\.venv\Scripts\python.exe" router.py
|
||||||
|
}
|
||||||
|
finally {
|
||||||
|
if ($litellm -and -not $litellm.HasExited) { Stop-Process -Id $litellm.Id -Force -ErrorAction SilentlyContinue }
|
||||||
|
Remove-Item ".router.pid" -ErrorAction SilentlyContinue
|
||||||
|
}
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
# Pull a model fleet into Ollama on Windows. Edit $models to taste.
|
||||||
|
# Names must line up with the routing map in router.py (see docs/ROUTING.md).
|
||||||
|
# Run: powershell -ExecutionPolicy Bypass -File .\scripts\pull-models.ps1
|
||||||
|
$models = @(
|
||||||
|
# general / small
|
||||||
|
"qwen3:8b", "qwen3:14b",
|
||||||
|
# reasoning / agentic
|
||||||
|
"qwen3.6:35b-a3b", "gpt-oss:20b", "glm-4.7-flash",
|
||||||
|
# coding
|
||||||
|
"qwen3-coder:30b",
|
||||||
|
# vision / OCR
|
||||||
|
"qwen3-vl:8b", "qwen3-vl:30b-a3b-instruct",
|
||||||
|
# multimodal (vision + audio) — gemma4:e4b is the default ROUTER_AUDIO_MODEL
|
||||||
|
"gemma4:e4b", "gemma4:12b", "gemma4:26b"
|
||||||
|
# example HF GGUF: "hf.co/USER/REPO:Q4_K_M"
|
||||||
|
)
|
||||||
|
|
||||||
|
$ok = @(); $fail = @(); $i = 0
|
||||||
|
foreach ($m in $models) {
|
||||||
|
$i++
|
||||||
|
Write-Host "=== [$i/$($models.Count)] pulling $m ==="
|
||||||
|
ollama pull $m
|
||||||
|
if ($LASTEXITCODE -eq 0) { $ok += $m } else { Write-Host "!! failed: $m"; $fail += $m }
|
||||||
|
}
|
||||||
|
Write-Host ""
|
||||||
|
Write-Host "done. $($ok.Count)/$($models.Count) succeeded."
|
||||||
|
if ($fail.Count) { Write-Host "failed: $($fail -join ', ')" }
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
# One-time setup on Windows: create the Python 3.12 venv and install LiteLLM + Pillow.
|
||||||
|
# The router itself is stdlib-only; this is for the LiteLLM backend.
|
||||||
|
# Run: powershell -ExecutionPolicy Bypass -File .\scripts\setup.ps1
|
||||||
|
$ErrorActionPreference = "Stop"
|
||||||
|
Set-Location (Split-Path $PSScriptRoot -Parent) # repo root
|
||||||
|
|
||||||
|
function Find-Py312 {
|
||||||
|
if (Get-Command py -ErrorAction SilentlyContinue) {
|
||||||
|
try { if ((& py -3.12 --version 2>&1) -match "3\.12") { return @("py", "-3.12") } } catch {}
|
||||||
|
}
|
||||||
|
foreach ($c in @("python3.12", "python")) {
|
||||||
|
if (Get-Command $c -ErrorAction SilentlyContinue) {
|
||||||
|
try { if ((& $c --version 2>&1) -match "3\.12") { return @($c) } } catch {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return $null
|
||||||
|
}
|
||||||
|
|
||||||
|
$py = Find-Py312
|
||||||
|
if (-not $py) {
|
||||||
|
Write-Error "Python 3.12 not found. Install it: winget install Python.Python.3.12 (or python.org)"
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
$exe = $py[0]
|
||||||
|
$baseArgs = if ($py.Count -gt 1) { $py[1..($py.Count - 1)] } else { @() }
|
||||||
|
Write-Host ">> using Python 3.12 via '$($py -join ' ')'"
|
||||||
|
|
||||||
|
if (-not (Test-Path ".venv")) { & $exe @baseArgs -m venv .venv }
|
||||||
|
$vpy = ".\.venv\Scripts\python.exe"
|
||||||
|
& $vpy -m pip install -q --upgrade pip
|
||||||
|
& $vpy -m pip install -q -r requirements.txt
|
||||||
|
& $vpy -c "import litellm, PIL; print('>> deps OK (litellm, pillow)')"
|
||||||
|
|
||||||
|
Write-Host ""
|
||||||
|
Write-Host "Next:"
|
||||||
|
Write-Host " .\scripts\pull-models.ps1 # pull a model fleet (edit the list first)"
|
||||||
|
Write-Host " .\run.ps1 # start LiteLLM + router"
|
||||||
Reference in New Issue
Block a user