Stale comment said 'Production unchanged until Ryan approves cutover' — router is now production. Replaced with accurate historical note.
58 lines
2.7 KiB
Django/Jinja
58 lines
2.7 KiB
Django/Jinja
[Unit]
|
|
Description=llama-server router — {{ llm_router_models_dir }} (OpenAI-compatible, port {{ llm_router_port }})
|
|
Documentation=https://github.com/ggml-org/llama.cpp
|
|
After=network.target nvidia-persistenced.service
|
|
Wants=nvidia-persistenced.service
|
|
|
|
[Service]
|
|
Type=simple
|
|
User={{ llm_service_user }}
|
|
Group={{ llm_service_user }}
|
|
Environment="HOME=/home/{{ llm_service_user }}"
|
|
ExecStart={{ llm_binary_path }} \
|
|
--models-dir {{ llm_router_models_dir }} \
|
|
--models-max {{ llm_router_models_max }} \
|
|
--host {{ llm_router_bind_address }} \
|
|
--port {{ llm_router_port }} \
|
|
--n-gpu-layers {{ llm_router_gpu_layers }} \
|
|
--ctx-size {{ llm_router_ctx_size }} \
|
|
--flash-attn {{ llm_router_flash_attn }} \
|
|
--cache-type-k {{ llm_router_cache_type_k }} \
|
|
--cache-type-v {{ llm_router_cache_type_v }} \
|
|
--batch-size {{ llm_router_batch_size }} \
|
|
--ubatch-size {{ llm_router_ubatch_size }} \
|
|
--parallel {{ llm_router_parallel }} \
|
|
--metrics
|
|
|
|
# ROUTER MODE NOTES (2026-08-12, t_0cca74a2):
|
|
# - NO -m/--model flag: this is what enables llama-server router/supervisor mode.
|
|
# Without -m, llama-server discovers all .gguf files in --models-dir, spawning
|
|
# each as its own child process on demand (LRU-eviction when over models-max).
|
|
# - --models-max {{ llm_router_models_max }} is HARDCODED TO 1.
|
|
# Default cap is 4 simultaneous — OOM on 24GB with a 20GB model.
|
|
# Do not increase without a VRAM budget review (see defaults/main.yml comment).
|
|
# - --models-dir /opt/models: auto-discovers all .gguf files. Keep that directory
|
|
# clean (Qwen-only) to avoid spurious extra entries in /v1/models.
|
|
# - Clients select a model via "model": "<gguf-basename-without-.gguf>" in their
|
|
# chat completion request. Hermes sends model: "<id>" on every request already.
|
|
# - Cold model load on first request: ~30-60s for Qwen3.6-35B. First response
|
|
# will be slow. This is expected. Document in runbook.
|
|
# - No --jinja flag: Qwen3.6-35B uses its own embedded chat template correctly.
|
|
# If per-model template overrides are ever needed, use --models-preset INI
|
|
# (but note GH #23460: sampler params in presets may not work in router mode).
|
|
#
|
|
# SHADOW DEPLOYMENT NOTE (historical — 2026-08-12, t_0cca74a2):
|
|
# This unit was originally deployed on port 8003 as a shadow. After validation,
|
|
# it was promoted to production on port 8002 (t_cd0d5388). The --port value
|
|
# above is the authoritative value; the port 8003 references below are historical.
|
|
# Production is now llama-server-router (this unit); llama-server-qwen is the rollback target.
|
|
Restart=on-failure
|
|
RestartSec=10
|
|
TimeoutStartSec=600
|
|
StandardOutput=journal
|
|
StandardError=journal
|
|
SyslogIdentifier=llama-server-router
|
|
|
|
[Install]
|
|
WantedBy=multi-user.target
|