[Unit] Description=vLLM OpenAI-compatible inference server — {{ item.id }} ({{ item.hf_repo }}) After=network-online.target nvidia-persistenced.service Wants=network-online.target nvidia-persistenced.service [Service] Type=simple User={{ vllm_venv_owner }} Group={{ vllm_venv_owner }} EnvironmentFile={{ vllm_api_key_env_file }} Environment="HOME=/home/{{ vllm_venv_owner }}" Environment="HF_HUB_CACHE={{ vllm_hf_hub_cache }}" Environment="HF_HOME={{ vllm_cache_dir }}" # vLLM's torch.compile path shells out to `ninja` by bare name (not via # venv-relative path) — without the venv's bin/ on PATH, systemd's minimal # default PATH causes FileNotFoundError: 'ninja' deep in compile, even # though `pip install vllm` installs the ninja package (and its console # script) INTO the venv. Caught during Phase 5 validation (t_ca1af9fb, # 2026-08-31): interactive SSH sessions have a different PATH than systemd # services, so this only reproduces under systemd, not manual testing. Environment="PATH={{ vllm_venv_path }}/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin" # FlashInfer's bundled sampling.cu JIT-compiles against a cub template API # (BlockAdjacentDifference::FlagHeads) that this flashinfer/CUDA toolkit # combination does not provide on RTX 3090 (SM86) — 100 compile errors, # confirmed upstream-known (vLLM GH #23023, #44305: FlashInfer sampler JIT # breaks on various SM targets across flashinfer/vLLM version combos). # Falls back to vLLM's native PyTorch sampler, which is fully supported and # only marginally slower for single-request/low-concurrency serving. Caught # during Phase 5 validation (t_ca1af9fb, 2026-08-31). Environment="VLLM_USE_FLASHINFER_SAMPLER=0" ExecStart={{ vllm_venv_path }}/bin/python -m vllm.entrypoints.openai.api_server \ --model {{ item.hf_repo }} \ --served-model-name {{ item.id }} \ --host {{ vllm_serve_host }} \ --port {{ item.port }} \ {% if item.role == 'embedding' %} --runner pooling \ --convert embed \ {% endif %} {% if item.trust_remote_code is defined and item.trust_remote_code %} --trust-remote-code \ {% endif %} {% if item.enforce_eager is defined and item.enforce_eager %} --enforce-eager \ {% endif %} {% if item.quantization is defined and item.quantization != 'none' %} --quantization {{ item.quantization }} \ {% endif %} {% if item.kv_cache_dtype is defined %} --kv-cache-dtype {{ item.kv_cache_dtype }} \ {% endif %} {% if item.kv_cache_memory_bytes is defined %} --kv-cache-memory-bytes {{ item.kv_cache_memory_bytes }} \ {% endif %} --gpu-memory-utilization {{ item.gpu_memory_utilization }} \ --max-model-len {{ item.max_model_len }} \ --dtype {{ vllm_dtype }} \ --api-key ${VLLM_API_KEY} \ {% if item.role != 'embedding' %} --enable-prefix-caching {% else %} --no-enable-prefix-caching {% endif %} Restart={{ vllm_restart_policy }} RestartSec=10 # vLLM torch.compile can take 4+ minutes before /health responds even after # weights are loaded (homelab-llm-inference skill pitfall) — give it room. TimeoutStartSec=600 StandardOutput=journal StandardError=journal SyslogIdentifier=vllm-{{ item.id }} [Install] WantedBy=multi-user.target