[Unit] Description=llama-server — Gemma 2 27B-it Q4_K_M (OpenAI-compatible inference) After=network.target nvidia-persistenced.service Wants=nvidia-persistenced.service [Service] Type=simple User={{ llm_venv_owner }} Group={{ llm_venv_owner }} Environment="HOME=/home/{{ llm_venv_owner }}" ExecStart=/opt/llama.cpp/build/bin/llama-server \ --model {{ llm_gguf_path }} \ --host {{ llm_serve_host }} \ --port {{ llm_serve_port }} \ --ctx-size {{ llm_max_model_len }} \ --n-gpu-layers {{ llm_gpu_layers }} \ --parallel {{ llm_parallel_slots }} \ --metrics # NOTE: no --chat-template flag — llama-server auto-detects and uses the # GGUF's own embedded Jinja chat template (verified correct Gemma-2 # start_of_turn/end_of_turn format for bartowski's gemma-2-27b-it-Q4_K_M). # The built-in "--chat-template gemma" name does NOT match this model's # expected format on this llama.cpp build and produced garbled completions. Restart=on-failure RestartSec=10 TimeoutStartSec=120 StandardOutput=journal StandardError=journal SyslogIdentifier=llama-server [Install] WantedBy=multi-user.target