- vllm.service.j2: branch on role==embedding for --runner pooling --convert embed, --no-enable-prefix-caching, per-model trust_remote_code toggle (needed for nomic-embed-text-v1.5's custom NomicBertModel code), and enforce_eager toggle (needed to avoid CUDA graph capture OOM when co-resident with another vLLM process on this 24GB card). - tasks/verify.yml: split completions vs embedding smoke tests -- embedding-mode instances don't serve /v1/completions. Assert a non-empty embedding vector, not just HTTP 200. - host_vars/astro-orbiter: enable nomic-embed-text-v1.5 (port 8020), lower primary model's gpu_memory_utilization 0.95->0.90 + add enforce_eager after finding 0.95 crash-looped 6-7x before stabilizing with co-resident nomic-embed (real fix, confirmed via NRestarts=0 after clean stop/start, not luck). - Hindsight (values.yaml + externalsecret.yaml): cut LLM + embeddings over to vLLM (:8000, :8020), wire the previously-unset HINDSIGHT_API_EMBEDDINGS_* env vars for the first time, and swap the API key secret source from the Nous fallback item to vllm/api-key (vLLM enforces real auth, llama-swap did not). - README: document the embedding-mode branch, VRAM findings, and a genuine architecture gap -- vLLM's one-model-per-process design cannot replace llama-swap's 5-model LRU roster on this 24GB card, so 21 Hermes profiles' aux-model consumers (Qwen3-8B-no_think, Phi-3.5-mini, Meta-Llama-3.1-8B, Qwen2.5-Coder-14B) and OpenViking's VLM stay on llama-swap. Full teardown (t_6dff1ecc) needs a human decision on the aux-model strategy before it can proceed.
71 lines
2.9 KiB
Django/Jinja
71 lines
2.9 KiB
Django/Jinja
[Unit]
|
|
Description=vLLM OpenAI-compatible inference server — {{ item.id }} ({{ item.hf_repo }})
|
|
After=network-online.target nvidia-persistenced.service
|
|
Wants=network-online.target nvidia-persistenced.service
|
|
|
|
[Service]
|
|
Type=simple
|
|
User={{ vllm_venv_owner }}
|
|
Group={{ vllm_venv_owner }}
|
|
EnvironmentFile={{ vllm_api_key_env_file }}
|
|
Environment="HOME=/home/{{ vllm_venv_owner }}"
|
|
Environment="HF_HUB_CACHE={{ vllm_hf_hub_cache }}"
|
|
Environment="HF_HOME={{ vllm_cache_dir }}"
|
|
# vLLM's torch.compile path shells out to `ninja` by bare name (not via
|
|
# venv-relative path) — without the venv's bin/ on PATH, systemd's minimal
|
|
# default PATH causes FileNotFoundError: 'ninja' deep in compile, even
|
|
# though `pip install vllm` installs the ninja package (and its console
|
|
# script) INTO the venv. Caught during Phase 5 validation (t_ca1af9fb,
|
|
# 2026-08-31): interactive SSH sessions have a different PATH than systemd
|
|
# services, so this only reproduces under systemd, not manual testing.
|
|
Environment="PATH={{ vllm_venv_path }}/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
|
|
# FlashInfer's bundled sampling.cu JIT-compiles against a cub template API
|
|
# (BlockAdjacentDifference::FlagHeads) that this flashinfer/CUDA toolkit
|
|
# combination does not provide on RTX 3090 (SM86) — 100 compile errors,
|
|
# confirmed upstream-known (vLLM GH #23023, #44305: FlashInfer sampler JIT
|
|
# breaks on various SM targets across flashinfer/vLLM version combos).
|
|
# Falls back to vLLM's native PyTorch sampler, which is fully supported and
|
|
# only marginally slower for single-request/low-concurrency serving. Caught
|
|
# during Phase 5 validation (t_ca1af9fb, 2026-08-31).
|
|
Environment="VLLM_USE_FLASHINFER_SAMPLER=0"
|
|
|
|
ExecStart={{ vllm_venv_path }}/bin/python -m vllm.entrypoints.openai.api_server \
|
|
--model {{ item.hf_repo }} \
|
|
--served-model-name {{ item.id }} \
|
|
--host {{ vllm_serve_host }} \
|
|
--port {{ item.port }} \
|
|
{% if item.role == 'embedding' %}
|
|
--runner pooling \
|
|
--convert embed \
|
|
{% endif %}
|
|
{% if item.trust_remote_code is defined and item.trust_remote_code %}
|
|
--trust-remote-code \
|
|
{% endif %}
|
|
{% if item.enforce_eager is defined and item.enforce_eager %}
|
|
--enforce-eager \
|
|
{% endif %}
|
|
{% if item.quantization is defined and item.quantization != 'none' %}
|
|
--quantization {{ item.quantization }} \
|
|
{% endif %}
|
|
--gpu-memory-utilization {{ item.gpu_memory_utilization }} \
|
|
--max-model-len {{ item.max_model_len }} \
|
|
--dtype {{ vllm_dtype }} \
|
|
--api-key ${VLLM_API_KEY} \
|
|
{% if item.role != 'embedding' %}
|
|
--enable-prefix-caching
|
|
{% else %}
|
|
--no-enable-prefix-caching
|
|
{% endif %}
|
|
|
|
Restart={{ vllm_restart_policy }}
|
|
RestartSec=10
|
|
# vLLM torch.compile can take 4+ minutes before /health responds even after
|
|
# weights are loaded (homelab-llm-inference skill pitfall) — give it room.
|
|
TimeoutStartSec=600
|
|
StandardOutput=journal
|
|
StandardError=journal
|
|
SyslogIdentifier=vllm-{{ item.id }}
|
|
|
|
[Install]
|
|
WantedBy=multi-user.target
|