Replace single llama-server with llama-swap so all benchmarked models are selectable from Hermes' menu and hot-swapped on the one P100. Menu: gpt-oss-20b (default, ~45s cold start), gemma-4-26b-a4b (MoE), gemma-4-12b, gemma-4-e4b. qwen3-30b-a3b excluded (OOMs at 64k in 16GB). All 64k, q8/q8 KV, --parallel 1. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
42 lines
1.5 KiB
YAML
42 lines
1.5 KiB
YAML
# llm stack — model-swapping LLM backend for the Hermes agent.
|
|
#
|
|
# llama-swap fronts multiple GGUF models on the single Tesla P100 (16GB). Only one
|
|
# model fits in VRAM at a time, so llama-swap presents all of them via /v1/models
|
|
# and hot-swaps on demand (selecting a different model = a few-second reload). The
|
|
# per-model llama-server commands + args live in llama-swap-config.yaml.
|
|
#
|
|
# The bundled llama.cpp in llama-swap:cuda is build 9803 (5c7c22c3e) — the same
|
|
# build validated on this Pascal card for gemma4 + gpt-oss. Default model and the
|
|
# selectable menu are driven from Hermes (~/.hermes/config.yaml: model.default =
|
|
# gpt-oss-20b; provider valhalla-p100 models: list = the keys in the swap config).
|
|
#
|
|
# Endpoint published on 172.20.0.1:8090 (edge bridge gateway, a host IP) for the
|
|
# host-side Hermes agent. Internal-only; no Caddy, no auth (LAN/host-only).
|
|
# Image is infra-pinned out of Watchtower.
|
|
services:
|
|
llama-swap:
|
|
image: ghcr.io/mostlygeek/llama-swap:cuda
|
|
container_name: llama-swap
|
|
restart: unless-stopped
|
|
labels:
|
|
- "com.centurylabs.watchtower.enable=false"
|
|
networks: [llm]
|
|
devices:
|
|
- "nvidia.com/gpu=0"
|
|
volumes:
|
|
- /storage1/labdata/llm/models:/models
|
|
- ./llama-swap-config.yaml:/app/config.yaml:ro
|
|
ports:
|
|
- "172.20.0.1:8090:8080"
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-fsS", "http://localhost:8080/health"]
|
|
interval: 30s
|
|
timeout: 10s
|
|
retries: 5
|
|
start_period: 30s
|
|
|
|
networks:
|
|
llm:
|
|
name: llm
|
|
driver: bridge
|