Files
homelab/ansible/roles/deploy-vllm/templates/vllm.service.j2
Hermes Agent service account 266b6c7be1 chore: apply all changes
2026-09-01 12:28:16 -05:00

86 lines
3.5 KiB
Django/Jinja

[Unit]
Description=vLLM OpenAI-compatible inference server — {{ item.id }} ({{ item.hf_repo }})
After=network-online.target nvidia-persistenced.service
Wants=network-online.target nvidia-persistenced.service
[Service]
Type=simple
User={{ vllm_venv_owner }}
Group={{ vllm_venv_owner }}
EnvironmentFile={{ vllm_api_key_env_file }}
Environment="HOME=/home/{{ vllm_venv_owner }}"
Environment="HF_HUB_CACHE={{ vllm_hf_hub_cache }}"
Environment="HF_HOME={{ vllm_cache_dir }}"
# vLLM's torch.compile path shells out to `ninja` by bare name (not via
# venv-relative path) — without the venv's bin/ on PATH, systemd's minimal
# default PATH causes FileNotFoundError: 'ninja' deep in compile, even
# though `pip install vllm` installs the ninja package (and its console
# script) INTO the venv. Caught during Phase 5 validation (t_ca1af9fb,
# 2026-08-31): interactive SSH sessions have a different PATH than systemd
# services, so this only reproduces under systemd, not manual testing.
Environment="PATH={{ vllm_venv_path }}/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
# FlashInfer's bundled sampling.cu JIT-compiles against a cub template API
# (BlockAdjacentDifference::FlagHeads) that this flashinfer/CUDA toolkit
# combination does not provide on RTX 3090 (SM86) — 100 compile errors,
# confirmed upstream-known (vLLM GH #23023, #44305: FlashInfer sampler JIT
# breaks on various SM targets across flashinfer/vLLM version combos).
# Falls back to vLLM's native PyTorch sampler, which is fully supported and
# only marginally slower for single-request/low-concurrency serving. Caught
# during Phase 5 validation (t_ca1af9fb, 2026-08-31).
Environment="VLLM_USE_FLASHINFER_SAMPLER=0"
ExecStart={{ vllm_venv_path }}/bin/python -m vllm.entrypoints.openai.api_server \
--model {{ item.hf_repo }} \
--served-model-name {{ item.id }} \
--host {{ vllm_serve_host }} \
--port {{ item.port }} \
{% if item.role == 'embedding' %}
--runner pooling \
--convert embed \
{% endif %}
{% if item.trust_remote_code is defined and item.trust_remote_code %}
--trust-remote-code \
{% endif %}
{% if item.enforce_eager is defined and item.enforce_eager %}
--enforce-eager \
{% endif %}
{% if item.enable_auto_tool_choice is defined and item.enable_auto_tool_choice %}
--enable-auto-tool-choice \
{% endif %}
{% if item.tool_call_parser is defined %}
--tool-call-parser {{ item.tool_call_parser }} \
{% endif %}
{% if item.reasoning_parser is defined %}
--reasoning-parser {{ item.reasoning_parser }} \
{% endif %}
{% if item.quantization is defined and item.quantization != 'none' %}
--quantization {{ item.quantization }} \
{% endif %}
{% if item.kv_cache_dtype is defined %}
--kv-cache-dtype {{ item.kv_cache_dtype }} \
{% endif %}
{% if item.kv_cache_memory_bytes is defined %}
--kv-cache-memory-bytes {{ item.kv_cache_memory_bytes }} \
{% endif %}
--gpu-memory-utilization {{ item.gpu_memory_utilization }} \
--max-model-len {{ item.max_model_len }} \
--dtype {{ vllm_dtype }} \
--api-key ${VLLM_API_KEY} \
{% if item.role != 'embedding' %}
--enable-prefix-caching
{% else %}
--no-enable-prefix-caching
{% endif %}
Restart={{ vllm_restart_policy }}
RestartSec=10
# vLLM torch.compile can take 4+ minutes before /health responds even after
# weights are loaded (homelab-llm-inference skill pitfall) — give it room.
TimeoutStartSec=600
StandardOutput=journal
StandardError=journal
SyslogIdentifier=vllm-{{ item.id }}
[Install]
WantedBy=multi-user.target